Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6b5685c5b3 | ||
|
|
52f44dd6d2 | ||
|
|
77c8f6c627 | ||
|
|
226b9debb7 | ||
|
|
6f35c61f93 | ||
|
|
638c0b7c50 | ||
|
|
9c4aa4ac1a |
No files matched your search
@@ -114,6 +114,115 @@ jobs:
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_torch_1_3:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,testing]
|
||||
- run: pip install torch==1.3.0
|
||||
- save_cache:
|
||||
key: v0.4-torch-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -s --make-reports=tests_torch ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_torch_1_4:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,testing]
|
||||
- run: pip install torch==1.4.0
|
||||
- save_cache:
|
||||
key: v0.4-torch-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -s --make-reports=tests_torch ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_torch_1_5:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,testing]
|
||||
- run: pip install torch==1.5.1
|
||||
- save_cache:
|
||||
key: v0.4-torch-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -s --make-reports=tests_torch ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_torch_1_6:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,testing]
|
||||
- run: pip install torch==1.6.0
|
||||
- save_cache:
|
||||
key: v0.4-torch-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -s --make-reports=tests_torch ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
|
||||
run_tests_tf:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
@@ -393,6 +502,10 @@ workflows:
|
||||
- run_tests_custom_tokenizers
|
||||
- run_tests_torch_and_tf
|
||||
- run_tests_torch
|
||||
- run_tests_torch_1_3
|
||||
- run_tests_torch_1_4
|
||||
- run_tests_torch_1_5
|
||||
- run_tests_torch_1_6
|
||||
- run_tests_tf
|
||||
- run_tests_flax
|
||||
- run_tests_pipelines_torch
|
||||
|
||||
@@ -36,13 +36,15 @@ assignees: ''
|
||||
examples/distillation: @VictorSanh
|
||||
nlp datasets: [different repo](https://github.com/huggingface/nlp)
|
||||
rust tokenizers: [different repo](https://github.com/huggingface/tokenizers)
|
||||
Text Generation: @TevenLeScao
|
||||
Text Generation: @patrickvonplaten @TevenLeScao
|
||||
blenderbot: @mariamabarham
|
||||
Bart: @sshleifer
|
||||
Marian: @sshleifer
|
||||
T5: @patrickvonplaten
|
||||
Longformer/Reformer: @patrickvonplaten
|
||||
TransfoXL/XLNet: @TevenLeScao
|
||||
TransfoXL/XLNet: @TevenLeScao
|
||||
RAG: @patrickvonplaten, @lhoestq
|
||||
FSTM: @stas00
|
||||
examples/seq2seq: @sshleifer
|
||||
examples/bert-loses-patience: @JetRunner
|
||||
tensorflow: @jplu
|
||||
|
||||
@@ -60,4 +60,5 @@ members/contributors which may be interested in your PR.
|
||||
tensorflow: @jplu
|
||||
examples/token-classification: @stefan-it
|
||||
documentation: @sgugger
|
||||
FSTM: @stas00
|
||||
-->
|
||||
+1
-1
@@ -68,7 +68,7 @@ For example for `run_glue`:
|
||||
|
||||
```bash
|
||||
python examples/xla_spawn.py --num_cores 8 \
|
||||
examples/text-classification/run_glue.py
|
||||
examples/text-classification/run_glue.py \
|
||||
--model_name_or_path bert-base-cased \
|
||||
--task_name mnli \
|
||||
--data_dir ./data/glue_data/MNLI \
|
||||
|
||||
@@ -65,7 +65,7 @@ class AlbertModelWithPabee(AlbertModel):
|
||||
|
||||
self.encoder = AlbertTransformerWithPabee(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
self.patience = 0
|
||||
self.inference_instances_num = 0
|
||||
self.inference_layers_num = 0
|
||||
@@ -228,7 +228,7 @@ class AlbertForSequenceClassificationWithPabee(AlbertPreTrainedModel):
|
||||
[nn.Linear(config.hidden_size, self.config.num_labels) for _ in range(config.num_hidden_layers)]
|
||||
)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ALBERT_INPUTS_DOCSTRING)
|
||||
def forward(
|
||||
|
||||
@@ -70,7 +70,7 @@ class BertModelWithPabee(BertModel):
|
||||
|
||||
self.encoder = BertEncoderWithPabee(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
self.patience = 0
|
||||
self.inference_instances_num = 0
|
||||
self.inference_layers_num = 0
|
||||
@@ -252,7 +252,7 @@ class BertForSequenceClassificationWithPabee(BertPreTrainedModel):
|
||||
[nn.Linear(config.hidden_size, self.config.num_labels) for _ in range(config.num_hidden_layers)]
|
||||
)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BERT_INPUTS_DOCSTRING)
|
||||
def forward(
|
||||
|
||||
@@ -115,7 +115,7 @@ class DeeBertModel(BertPreTrainedModel):
|
||||
self.encoder = DeeBertEncoder(config)
|
||||
self.pooler = BertPooler(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def init_highway_pooler(self):
|
||||
self.encoder.init_highway_pooler(self.pooler)
|
||||
@@ -286,7 +286,7 @@ class DeeBertForSequenceClassification(BertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, self.config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BERT_INPUTS_DOCSTRING)
|
||||
def forward(
|
||||
|
||||
@@ -23,7 +23,7 @@ class DeeRobertaModel(DeeBertModel):
|
||||
super().__init__(config)
|
||||
|
||||
self.embeddings = RobertaEmbeddings(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
|
||||
@@ -264,7 +264,15 @@ def main():
|
||||
def tokenize_function(examples):
|
||||
# Remove empty lines
|
||||
examples["text"] = [line for line in examples["text"] if len(line) > 0 and not line.isspace()]
|
||||
return tokenizer(examples["text"], padding=padding, truncation=True, max_length=data_args.max_seq_length)
|
||||
return tokenizer(
|
||||
examples["text"],
|
||||
padding=padding,
|
||||
truncation=True,
|
||||
max_length=data_args.max_seq_length,
|
||||
# We use this option because DataCollatorForLanguageModeling (see below) is more efficient when it
|
||||
# receives the `special_tokens_mask`.
|
||||
return_special_tokens_mask=True,
|
||||
)
|
||||
|
||||
tokenized_datasets = datasets.map(
|
||||
tokenize_function,
|
||||
@@ -275,8 +283,10 @@ def main():
|
||||
)
|
||||
else:
|
||||
# Otherwise, we tokenize every text, then concatenate them together before splitting them in smaller parts.
|
||||
# We use `return_special_tokens_mask=True` because DataCollatorForLanguageModeling (see below) is more
|
||||
# efficient when it receives the `special_tokens_mask`.
|
||||
def tokenize_function(examples):
|
||||
return tokenizer(examples[text_column_name])
|
||||
return tokenizer(examples[text_column_name], return_special_tokens_mask=True)
|
||||
|
||||
tokenized_datasets = datasets.map(
|
||||
tokenize_function,
|
||||
|
||||
@@ -482,7 +482,7 @@ class MaskedBertModel(MaskedBertPreTrainedModel):
|
||||
self.encoder = BertEncoder(config)
|
||||
self.pooler = BertPooler(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -669,7 +669,7 @@ class MaskedBertForSequenceClassification(MaskedBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, self.config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MASKED_BERT_INPUTS_DOCSTRING)
|
||||
def forward(
|
||||
@@ -754,7 +754,7 @@ class MaskedBertForMultipleChoice(MaskedBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MASKED_BERT_INPUTS_DOCSTRING)
|
||||
def forward(
|
||||
@@ -844,7 +844,7 @@ class MaskedBertForTokenClassification(MaskedBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MASKED_BERT_INPUTS_DOCSTRING)
|
||||
def forward(
|
||||
@@ -930,7 +930,7 @@ class MaskedBertForQuestionAnswering(MaskedBertPreTrainedModel):
|
||||
self.bert = MaskedBertModel(config)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MASKED_BERT_INPUTS_DOCSTRING)
|
||||
def forward(
|
||||
|
||||
@@ -66,14 +66,16 @@ class Split(Enum):
|
||||
|
||||
|
||||
class TokenClassificationTask:
|
||||
def read_examples_from_file(self, data_dir, mode: Union[Split, str]) -> List[InputExample]:
|
||||
@staticmethod
|
||||
def read_examples_from_file(data_dir, mode: Union[Split, str]) -> List[InputExample]:
|
||||
raise NotImplementedError
|
||||
|
||||
def get_labels(self, path: str) -> List[str]:
|
||||
@staticmethod
|
||||
def get_labels(path: str) -> List[str]:
|
||||
raise NotImplementedError
|
||||
|
||||
@staticmethod
|
||||
def convert_examples_to_features(
|
||||
self,
|
||||
examples: List[InputExample],
|
||||
label_list: List[str],
|
||||
max_seq_length: int,
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
---
|
||||
language:
|
||||
- en
|
||||
tags:
|
||||
- bluebert
|
||||
license:
|
||||
- PUBLIC DOMAIN NOTICE
|
||||
datasets:
|
||||
- pubmed
|
||||
|
||||
---
|
||||
|
||||
# BlueBert-Base, Uncased, PubMed
|
||||
|
||||
## Model description
|
||||
|
||||
A BERT model pre-trained on PubMed abstracts
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
#### How to use
|
||||
|
||||
Please see https://github.com/ncbi-nlp/bluebert
|
||||
|
||||
## Training data
|
||||
|
||||
We provide [preprocessed PubMed texts](https://ftp.ncbi.nlm.nih.gov/pub/lu/Suppl/NCBI-BERT/pubmed_uncased_sentence_nltk.txt.tar.gz) that were used to pre-train the BlueBERT models.
|
||||
The corpus contains ~4000M words extracted from the [PubMed ASCII code version](https://www.ncbi.nlm.nih.gov/research/bionlp/APIs/BioC-PubMed/).
|
||||
|
||||
Pre-trained model: https://huggingface.co/bert-base-uncased
|
||||
|
||||
## Training procedure
|
||||
|
||||
* lowercasing the text
|
||||
* removing speical chars `\x00`-`\x7F`
|
||||
* tokenizing the text using the [NLTK Treebank tokenizer](https://www.nltk.org/_modules/nltk/tokenize/treebank.html)
|
||||
|
||||
Below is a code snippet for more details.
|
||||
|
||||
```python
|
||||
value = value.lower()
|
||||
value = re.sub(r'[\r\n]+', ' ', value)
|
||||
value = re.sub(r'[^\x00-\x7F]+', ' ', value)
|
||||
|
||||
tokenized = TreebankWordTokenizer().tokenize(value)
|
||||
sentence = ' '.join(tokenized)
|
||||
sentence = re.sub(r"\s's\b", "'s", sentence)
|
||||
```
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@InProceedings{peng2019transfer,
|
||||
author = {Yifan Peng and Shankai Yan and Zhiyong Lu},
|
||||
title = {Transfer Learning in Biomedical Natural Language Processing: An Evaluation of BERT and ELMo on Ten Benchmarking Datasets},
|
||||
booktitle = {Proceedings of the 2019 Workshop on Biomedical Natural Language Processing (BioNLP 2019)},
|
||||
year = {2019},
|
||||
pages = {58--65},
|
||||
}
|
||||
```
|
||||
@@ -281,7 +281,6 @@ if is_torch_available():
|
||||
from .data.data_collator import (
|
||||
DataCollator,
|
||||
DataCollatorForLanguageModeling,
|
||||
DataCollatorForNextSentencePrediction,
|
||||
DataCollatorForPermutationLanguageModeling,
|
||||
DataCollatorForSOP,
|
||||
DataCollatorForTokenClassification,
|
||||
|
||||
@@ -254,7 +254,6 @@ class PyTorchBenchmark(Benchmark):
|
||||
else:
|
||||
# cpu
|
||||
memory_bytes = measure_peak_memory_cpu(func)
|
||||
print("PEAK", memory_bytes)
|
||||
memory = Memory(memory_bytes) if isinstance(memory_bytes, int) else memory_bytes
|
||||
|
||||
if self.args.trace_memory_line_by_line:
|
||||
@@ -262,7 +261,6 @@ class PyTorchBenchmark(Benchmark):
|
||||
else:
|
||||
summary = None
|
||||
|
||||
print(memory, summary)
|
||||
return memory, summary
|
||||
except RuntimeError as e:
|
||||
self.print_fn("Doesn't fit on GPU. {}".format(e))
|
||||
|
||||
@@ -76,7 +76,12 @@ def separate_process_wrapper_fn(func: Callable[[], None], do_multi_processing: b
|
||||
# run function in an individual
|
||||
# process to get correct memory
|
||||
def wrapper_func(queue: Queue, *args):
|
||||
result = func(*args)
|
||||
try:
|
||||
result = func(*args)
|
||||
except Exception as e:
|
||||
logger.error(e)
|
||||
print(e)
|
||||
result = "N/A"
|
||||
queue.put(result)
|
||||
|
||||
queue = Queue()
|
||||
@@ -286,13 +291,13 @@ def measure_peak_memory_cpu(function: Callable[[], None], interval=0.5, device_i
|
||||
# receive memory and num measurements
|
||||
max_memory = parent_connection.recv()
|
||||
num_measurements = parent_connection.recv()
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
# kill process in a clean way
|
||||
parent = psutil.Process(os.getpid())
|
||||
for child in parent.children(recursive=True):
|
||||
os.kill(child.pid, SIGKILL)
|
||||
mem_process.join(0)
|
||||
raise RuntimeError(f"Process killed. Error in Process: {e}")
|
||||
raise RuntimeError("Process killed. Error in Process")
|
||||
|
||||
# run process at least 20 * interval or until it finishes
|
||||
mem_process.join(20 * interval)
|
||||
@@ -687,8 +692,7 @@ class Benchmark(ABC):
|
||||
for sequence_length in self.args.sequence_lengths:
|
||||
if self.args.inference:
|
||||
if self.args.memory:
|
||||
outputs = self.inference_memory(model_name, batch_size, sequence_length)
|
||||
memory, inference_summary = outputs if type(outputs) == tuple else (outputs, None)
|
||||
memory, inference_summary = self.inference_memory(model_name, batch_size, sequence_length)
|
||||
inference_result_memory[model_name]["result"][batch_size][sequence_length] = memory
|
||||
if self.args.speed:
|
||||
time = self.inference_speed(model_name, batch_size, sequence_length)
|
||||
@@ -696,8 +700,7 @@ class Benchmark(ABC):
|
||||
|
||||
if self.args.training:
|
||||
if self.args.memory:
|
||||
outputs = self.train_memory(model_name, batch_size, sequence_length)
|
||||
memory, train_summary = outputs if type(outputs) == tuple else (outputs, None)
|
||||
memory, train_summary = self.train_memory(model_name, batch_size, sequence_length)
|
||||
train_result_memory[model_name]["result"][batch_size][sequence_length] = memory
|
||||
if self.args.speed:
|
||||
time = self.train_speed(model_name, batch_size, sequence_length)
|
||||
|
||||
@@ -170,7 +170,6 @@ class PretrainedConfig(object):
|
||||
self.torchscript = kwargs.pop("torchscript", False) # Only used by PyTorch models
|
||||
self.use_bfloat16 = kwargs.pop("use_bfloat16", False)
|
||||
self.pruned_heads = kwargs.pop("pruned_heads", {})
|
||||
self.gradient_checkpointing = kwargs.pop("gradient_checkpointing", False)
|
||||
self.tie_word_embeddings = kwargs.pop(
|
||||
"tie_word_embeddings", True
|
||||
) # Whether input and output word embeddings should be tied for all MLM, LM and Seq2Seq models.
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import random
|
||||
import warnings
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable, Dict, List, NewType, Optional, Tuple, Union
|
||||
|
||||
@@ -175,72 +176,111 @@ class DataCollatorForTokenClassification:
|
||||
return batch
|
||||
|
||||
|
||||
def _collate_batch(examples, tokenizer):
|
||||
"""Collate `examples` into a batch, using the information in `tokenizer` for padding if necessary."""
|
||||
# Tensorize if necessary.
|
||||
if isinstance(examples[0], (list, tuple)):
|
||||
examples = [torch.tensor(e, dtype=torch.long) for e in examples]
|
||||
|
||||
# Check if padding is necessary.
|
||||
length_of_first = examples[0].size(0)
|
||||
are_tensors_same_length = all(x.size(0) == length_of_first for x in examples)
|
||||
if are_tensors_same_length:
|
||||
return torch.stack(examples, dim=0)
|
||||
|
||||
# If yes, check if we have a `pad_token`.
|
||||
if tokenizer._pad_token is None:
|
||||
raise ValueError(
|
||||
"You are attempting to pad samples but the tokenizer you are using"
|
||||
f" ({tokenizer.__class__.__name__}) does not have a pad token."
|
||||
)
|
||||
|
||||
# Creating the full tensor and filling it with our data.
|
||||
max_length = max(x.size(0) for x in examples)
|
||||
result = examples[0].new_full([len(examples), max_length], tokenizer.pad_token_id)
|
||||
for i, example in enumerate(examples):
|
||||
if tokenizer.padding_side == "right":
|
||||
result[i, : example.shape[0]] = example
|
||||
else:
|
||||
result[i, -example.shape[0] :] = example
|
||||
return result
|
||||
|
||||
|
||||
@dataclass
|
||||
class DataCollatorForLanguageModeling:
|
||||
"""
|
||||
Data collator used for language modeling.
|
||||
Data collator used for language modeling. Inputs are dynamically padded to the maximum length of a batch if they
|
||||
are not all of the same length.
|
||||
|
||||
- collates batches of tensors, honoring their tokenizer's pad_token
|
||||
- preprocesses batches for masked language modeling
|
||||
Args:
|
||||
tokenizer (:class:`~transformers.PreTrainedTokenizer` or :class:`~transformers.PreTrainedTokenizerFast`):
|
||||
The tokenizer used for encoding the data.
|
||||
mlm (:obj:`bool`, `optional`, defaults to :obj:`True`):
|
||||
Whether or not to use masked language modeling. If set to :obj:`False`, the labels are the same as the
|
||||
inputs with the padding tokens ignored (by setting them to -100). Otherwise, the labels are -100 for
|
||||
non-masked tokens and the value to predict for the masked token.
|
||||
mlm_probability (:obj:`float`, `optional`, defaults to 0.15):
|
||||
The probability with which to (randomly) mask tokens in the input, when :obj:`mlm` is set to :obj:`True`.
|
||||
|
||||
.. note::
|
||||
|
||||
For best performance, this data collator should be used with a dataset having items that are dictionaries or
|
||||
BatchEncoding, with the :obj:`"special_tokens_mask"` key, as returned by a
|
||||
:class:`~transformers.PreTrainedTokenizer` or a :class:`~transformers.PreTrainedTokenizerFast` with the
|
||||
argument :obj:`return_special_tokens_mask=True`.
|
||||
"""
|
||||
|
||||
tokenizer: PreTrainedTokenizerBase
|
||||
mlm: bool = True
|
||||
mlm_probability: float = 0.15
|
||||
|
||||
def __post_init__(self):
|
||||
if self.mlm and self.tokenizer.mask_token is None:
|
||||
raise ValueError(
|
||||
"This tokenizer does not have a mask token which is necessary for masked language modeling. "
|
||||
"You should pass `mlm=False` to train on causal language modeling instead."
|
||||
)
|
||||
|
||||
def __call__(
|
||||
self, examples: List[Union[List[int], torch.Tensor, Dict[str, torch.Tensor]]]
|
||||
) -> Dict[str, torch.Tensor]:
|
||||
# Handle dict or lists with proper padding and conversion to tensor.
|
||||
if isinstance(examples[0], (dict, BatchEncoding)):
|
||||
examples = [e["input_ids"] for e in examples]
|
||||
batch = self._tensorize_batch(examples)
|
||||
if self.mlm:
|
||||
inputs, labels = self.mask_tokens(batch)
|
||||
return {"input_ids": inputs, "labels": labels}
|
||||
batch = self.tokenizer.pad(examples, return_tensors="pt")
|
||||
else:
|
||||
labels = batch.clone().detach()
|
||||
batch = {"input_ids": _collate_batch(examples, self.tokenizer)}
|
||||
|
||||
# If special token mask has been preprocessed, pop it from the dict.
|
||||
special_tokens_mask = batch.pop("special_tokens_mask", None)
|
||||
if self.mlm:
|
||||
batch["input_ids"], batch["labels"] = self.mask_tokens(
|
||||
batch["input_ids"], special_tokens_mask=special_tokens_mask
|
||||
)
|
||||
else:
|
||||
labels = batch["input_ids"]
|
||||
if self.tokenizer.pad_token_id is not None:
|
||||
labels[labels == self.tokenizer.pad_token_id] = -100
|
||||
return {"input_ids": batch, "labels": labels}
|
||||
batch["labels"] = labels
|
||||
return batch
|
||||
|
||||
def _tensorize_batch(
|
||||
self, examples: List[Union[List[int], torch.Tensor, Dict[str, torch.Tensor]]]
|
||||
) -> torch.Tensor:
|
||||
# In order to accept both lists of lists and lists of Tensors
|
||||
if isinstance(examples[0], (list, tuple)):
|
||||
examples = [torch.tensor(e, dtype=torch.long) for e in examples]
|
||||
length_of_first = examples[0].size(0)
|
||||
are_tensors_same_length = all(x.size(0) == length_of_first for x in examples)
|
||||
if are_tensors_same_length:
|
||||
return torch.stack(examples, dim=0)
|
||||
else:
|
||||
if self.tokenizer._pad_token is None:
|
||||
raise ValueError(
|
||||
"You are attempting to pad samples but the tokenizer you are using"
|
||||
f" ({self.tokenizer.__class__.__name__}) does not have one."
|
||||
)
|
||||
return pad_sequence(examples, batch_first=True, padding_value=self.tokenizer.pad_token_id)
|
||||
|
||||
def mask_tokens(self, inputs: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
def mask_tokens(
|
||||
self, inputs: torch.Tensor, special_tokens_mask: Optional[torch.Tensor] = None
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
"""
|
||||
Prepare masked tokens inputs/labels for masked language modeling: 80% MASK, 10% random, 10% original.
|
||||
"""
|
||||
|
||||
if self.tokenizer.mask_token is None:
|
||||
raise ValueError(
|
||||
"This tokenizer does not have a mask token which is necessary for masked language modeling. Remove the --mlm flag if you want to use this tokenizer."
|
||||
)
|
||||
|
||||
labels = inputs.clone()
|
||||
# We sample a few tokens in each sequence for masked-LM training (with probability args.mlm_probability defaults to 0.15 in Bert/RoBERTa)
|
||||
# We sample a few tokens in each sequence for MLM training (with probability `self.mlm_probability`)
|
||||
probability_matrix = torch.full(labels.shape, self.mlm_probability)
|
||||
special_tokens_mask = [
|
||||
self.tokenizer.get_special_tokens_mask(val, already_has_special_tokens=True) for val in labels.tolist()
|
||||
]
|
||||
probability_matrix.masked_fill_(torch.tensor(special_tokens_mask, dtype=torch.bool), value=0.0)
|
||||
if self.tokenizer._pad_token is not None:
|
||||
padding_mask = labels.eq(self.tokenizer.pad_token_id)
|
||||
probability_matrix.masked_fill_(padding_mask, value=0.0)
|
||||
if special_tokens_mask is None:
|
||||
special_tokens_mask = [
|
||||
self.tokenizer.get_special_tokens_mask(val, already_has_special_tokens=True) for val in labels.tolist()
|
||||
]
|
||||
special_tokens_mask = torch.tensor(special_tokens_mask, dtype=torch.bool)
|
||||
else:
|
||||
special_tokens_mask = special_tokens_mask.bool()
|
||||
|
||||
probability_matrix.masked_fill_(special_tokens_mask, value=0.0)
|
||||
masked_indices = torch.bernoulli(probability_matrix).bool()
|
||||
labels[~masked_indices] = -100 # We only compute loss on masked tokens
|
||||
|
||||
@@ -385,9 +425,16 @@ class DataCollatorForSOP(DataCollatorForLanguageModeling):
|
||||
- preprocesses batches for both masked language modeling and sentence order prediction
|
||||
"""
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
warnings.warn(
|
||||
"DataCollatorForSOP is deprecated and will be removed in a future version, you can now use "
|
||||
"DataCollatorForLanguageModeling instead.",
|
||||
FutureWarning,
|
||||
)
|
||||
|
||||
def __call__(self, examples: List[Dict[str, torch.Tensor]]) -> Dict[str, torch.Tensor]:
|
||||
input_ids = [example["input_ids"] for example in examples]
|
||||
input_ids = self._tensorize_batch(input_ids)
|
||||
input_ids = _collate_batch(input_ids, self.tokenizer)
|
||||
input_ids, labels, attention_mask = self.mask_tokens(input_ids)
|
||||
|
||||
token_type_ids = [example["token_type_ids"] for example in examples]
|
||||
@@ -582,136 +629,3 @@ class DataCollatorForPermutationLanguageModeling:
|
||||
) & masked_indices[i]
|
||||
|
||||
return inputs.long(), perm_mask, target_mapping, labels.long()
|
||||
|
||||
|
||||
@dataclass
|
||||
class DataCollatorForNextSentencePrediction:
|
||||
"""
|
||||
Data collator used for next sentence prediction. - collates examples which contains pre-generated negative examples
|
||||
- preprocesses batches for masked language modeling
|
||||
"""
|
||||
|
||||
tokenizer: PreTrainedTokenizerBase
|
||||
mlm: bool = True
|
||||
block_size: int = 512
|
||||
short_seq_probability: float = 0.1
|
||||
nsp_probability: float = 0.5
|
||||
mlm_probability: float = 0.15
|
||||
|
||||
def __call__(self, examples: List[Dict[str, torch.Tensor]]) -> Dict[str, torch.Tensor]:
|
||||
"""
|
||||
The input should contain negative examples, :class:`~transformers.DataCollatorForNextSentencePrediction` will
|
||||
not generate any negative examples
|
||||
|
||||
Args:
|
||||
examples (:obj:`List[Dict]`): Each dictionary should have the following keys:
|
||||
|
||||
- ``tokens_a``: A sequence of tokens, which should appear before ``tokens_b`` in the text.
|
||||
- ``tokens_b``: A sequence of tokens, which should appear after ``tokens_a`` in the text.
|
||||
- ``is_random_next``: 1 if this pair is generated randomly, else 0.
|
||||
"""
|
||||
|
||||
tokens_a = [e["tokens_a"] for e in examples]
|
||||
tokens_b = [e["tokens_b"] for e in examples]
|
||||
nsp_labels = [1 if e["is_random_next"] else 0 for e in examples]
|
||||
|
||||
input_ids = []
|
||||
segment_ids = []
|
||||
attention_masks = []
|
||||
|
||||
assert len(tokens_a) == len(tokens_b)
|
||||
for i in range(len(tokens_a)):
|
||||
input_id, attention_mask, segment_id = self.create_features_from_example(tokens_a[i], tokens_b[i])
|
||||
input_ids.append(input_id)
|
||||
segment_ids.append(segment_id)
|
||||
attention_masks.append(attention_mask)
|
||||
if self.mlm:
|
||||
input_ids, mlm_labels = self.mask_tokens(self._tensorize_batch(input_ids))
|
||||
else:
|
||||
input_ids = self._tensorize_batch(input_ids)
|
||||
|
||||
result = {
|
||||
"input_ids": input_ids,
|
||||
"attention_mask": self._tensorize_batch(attention_masks),
|
||||
"token_type_ids": self._tensorize_batch(segment_ids),
|
||||
"labels": mlm_labels if self.mlm else None,
|
||||
"next_sentence_label": torch.tensor(nsp_labels),
|
||||
}
|
||||
return result
|
||||
|
||||
def _tensorize_batch(self, examples: List[torch.Tensor]) -> torch.Tensor:
|
||||
length_of_first = examples[0].size(0)
|
||||
are_tensors_same_length = all(x.size(0) == length_of_first for x in examples)
|
||||
if are_tensors_same_length:
|
||||
return torch.stack(examples, dim=0)
|
||||
else:
|
||||
if self.tokenizer._pad_token is None:
|
||||
raise ValueError(
|
||||
"You are attempting to pad samples but the tokenizer you are using"
|
||||
f" ({self.tokenizer.__class__.__name__}) does not have one."
|
||||
)
|
||||
return pad_sequence(examples, batch_first=True, padding_value=self.tokenizer.pad_token_id)
|
||||
|
||||
def create_features_from_example(self, tokens_a, tokens_b):
|
||||
"""Creates examples for a single document."""
|
||||
|
||||
max_num_tokens = self.block_size - self.tokenizer.num_special_tokens_to_add(pair=True)
|
||||
|
||||
tokens_a, tokens_b, _ = self.tokenizer.truncate_sequences(
|
||||
tokens_a,
|
||||
tokens_b,
|
||||
num_tokens_to_remove=len(tokens_a) + len(tokens_b) - max_num_tokens,
|
||||
truncation_strategy="longest_first",
|
||||
)
|
||||
|
||||
input_id = self.tokenizer.build_inputs_with_special_tokens(tokens_a, tokens_b)
|
||||
attention_mask = [1] * len(input_id)
|
||||
segment_id = self.tokenizer.create_token_type_ids_from_sequences(tokens_a, tokens_b)
|
||||
assert len(input_id) <= self.block_size
|
||||
|
||||
# pad
|
||||
while len(input_id) < self.block_size:
|
||||
input_id.append(0)
|
||||
attention_mask.append(0)
|
||||
segment_id.append(0)
|
||||
|
||||
input_id = torch.tensor(input_id)
|
||||
attention_mask = torch.tensor(attention_mask)
|
||||
segment_id = torch.tensor(segment_id)
|
||||
|
||||
return input_id, attention_mask, segment_id
|
||||
|
||||
def mask_tokens(self, inputs: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
"""
|
||||
Prepare masked tokens inputs/labels for masked language modeling: 80% MASK, 10% random, 10% original.
|
||||
"""
|
||||
|
||||
if self.tokenizer.mask_token is None:
|
||||
raise ValueError(
|
||||
"This tokenizer does not have a mask token which is necessary for masked language modeling. Remove the --mlm flag if you want to use this tokenizer."
|
||||
)
|
||||
|
||||
labels = inputs.clone()
|
||||
# We sample a few tokens in each sequence for masked-LM training (with probability args.mlm_probability defaults to 0.15 in Bert/RoBERTa)
|
||||
probability_matrix = torch.full(labels.shape, self.mlm_probability)
|
||||
special_tokens_mask = [
|
||||
self.tokenizer.get_special_tokens_mask(val, already_has_special_tokens=True) for val in labels.tolist()
|
||||
]
|
||||
probability_matrix.masked_fill_(torch.tensor(special_tokens_mask, dtype=torch.bool), value=0.0)
|
||||
if self.tokenizer._pad_token is not None:
|
||||
padding_mask = labels.eq(self.tokenizer.pad_token_id)
|
||||
probability_matrix.masked_fill_(padding_mask, value=0.0)
|
||||
masked_indices = torch.bernoulli(probability_matrix).bool()
|
||||
labels[~masked_indices] = -100 # We only compute loss on masked tokens
|
||||
|
||||
# 80% of the time, we replace masked input tokens with tokenizer.mask_token ([MASK])
|
||||
indices_replaced = torch.bernoulli(torch.full(labels.shape, 0.8)).bool() & masked_indices
|
||||
inputs[indices_replaced] = self.tokenizer.convert_tokens_to_ids(self.tokenizer.mask_token)
|
||||
|
||||
# 10% of the time, we replace masked input tokens with random word
|
||||
indices_random = torch.bernoulli(torch.full(labels.shape, 0.5)).bool() & masked_indices & ~indices_replaced
|
||||
random_words = torch.randint(len(self.tokenizer), labels.shape, dtype=torch.long)
|
||||
inputs[indices_random] = random_words[indices_random]
|
||||
|
||||
# The rest of the time (10% of the time) we keep the masked input tokens unchanged
|
||||
return inputs, labels
|
||||
@@ -3,6 +3,7 @@ import os
|
||||
import pickle
|
||||
import random
|
||||
import time
|
||||
import warnings
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import torch
|
||||
@@ -17,6 +18,11 @@ from ...utils import logging
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
|
||||
DEPRECATION_WARNING = (
|
||||
"This dataset will be removed from the library soon, preprocessing should be handled with the 🤗 Datasets library."
|
||||
)
|
||||
|
||||
|
||||
class TextDataset(Dataset):
|
||||
"""
|
||||
This will be superseded by a framework-agnostic approach soon.
|
||||
@@ -30,6 +36,7 @@ class TextDataset(Dataset):
|
||||
overwrite_cache=False,
|
||||
cache_dir: Optional[str] = None,
|
||||
):
|
||||
warnings.warn(DEPRECATION_WARNING, FutureWarning)
|
||||
assert os.path.isfile(file_path), f"Input file path {file_path} not found"
|
||||
|
||||
block_size = block_size - tokenizer.num_special_tokens_to_add(pair=False)
|
||||
@@ -94,6 +101,7 @@ class LineByLineTextDataset(Dataset):
|
||||
"""
|
||||
|
||||
def __init__(self, tokenizer: PreTrainedTokenizer, file_path: str, block_size: int):
|
||||
warnings.warn(DEPRECATION_WARNING, FutureWarning)
|
||||
assert os.path.isfile(file_path), f"Input file path {file_path} not found"
|
||||
# Here, we do not cache the features, operating under the assumption
|
||||
# that we will soon use fast multithreaded tokenizers from the
|
||||
@@ -120,6 +128,7 @@ class LineByLineWithRefDataset(Dataset):
|
||||
"""
|
||||
|
||||
def __init__(self, tokenizer: PreTrainedTokenizer, file_path: str, block_size: int, ref_path: str):
|
||||
warnings.warn(DEPRECATION_WARNING, FutureWarning)
|
||||
assert os.path.isfile(file_path), f"Input file path {file_path} not found"
|
||||
assert os.path.isfile(ref_path), f"Ref file path {file_path} not found"
|
||||
# Here, we do not cache the features, operating under the assumption
|
||||
@@ -156,6 +165,7 @@ class LineByLineWithSOPTextDataset(Dataset):
|
||||
"""
|
||||
|
||||
def __init__(self, tokenizer: PreTrainedTokenizer, file_dir: str, block_size: int):
|
||||
warnings.warn(DEPRECATION_WARNING, FutureWarning)
|
||||
assert os.path.isdir(file_dir)
|
||||
logger.info(f"Creating features from dataset file folder at {file_dir}")
|
||||
self.examples = []
|
||||
@@ -305,6 +315,7 @@ class TextDatasetForNextSentencePrediction(Dataset):
|
||||
short_seq_probability=0.1,
|
||||
nsp_probability=0.5,
|
||||
):
|
||||
warnings.warn(DEPRECATION_WARNING, FutureWarning)
|
||||
assert os.path.isfile(file_path), f"Input file path {file_path} not found"
|
||||
|
||||
self.block_size = block_size - tokenizer.num_special_tokens_to_add(pair=True)
|
||||
@@ -449,9 +460,18 @@ class TextDatasetForNextSentencePrediction(Dataset):
|
||||
assert len(tokens_a) >= 1
|
||||
assert len(tokens_b) >= 1
|
||||
|
||||
self.examples.append(
|
||||
{"tokens_a": tokens_a, "tokens_b": tokens_b, "is_random_next": is_random_next}
|
||||
)
|
||||
# add special tokens
|
||||
input_ids = self.tokenizer.build_inputs_with_special_tokens(tokens_a, tokens_b)
|
||||
# add token type ids, 0 for sentence a, 1 for sentence b
|
||||
token_type_ids = self.tokenizer.create_token_type_ids_from_sequences(tokens_a, tokens_b)
|
||||
|
||||
example = {
|
||||
"input_ids": torch.tensor(input_ids, dtype=torch.long),
|
||||
"token_type_ids": torch.tensor(token_type_ids, dtype=torch.long),
|
||||
"next_sentence_label": torch.tensor(1 if is_random_next else 0, dtype=torch.long),
|
||||
}
|
||||
|
||||
self.examples.append(example)
|
||||
|
||||
current_chunk = []
|
||||
current_length = 0
|
||||
|
||||
@@ -600,7 +600,7 @@ class AlbertModel(AlbertPreTrainedModel):
|
||||
self.pooler = None
|
||||
self.pooler_activation = None
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -608,9 +608,6 @@ class AlbertModel(AlbertPreTrainedModel):
|
||||
def set_input_embeddings(self, value):
|
||||
self.embeddings.word_embeddings = value
|
||||
|
||||
def get_layers(self):
|
||||
return [layer for layer_group in self.encoder.albert_layer_groups for layer in layer_group.albert_layers]
|
||||
|
||||
def _resize_token_embeddings(self, new_num_tokens):
|
||||
old_embeddings = self.embeddings.word_embeddings
|
||||
new_embeddings = self._get_resized_embeddings(old_embeddings, new_num_tokens)
|
||||
@@ -653,13 +650,11 @@ class AlbertModel(AlbertPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -724,7 +719,7 @@ class AlbertForPreTraining(AlbertPreTrainedModel):
|
||||
self.predictions = AlbertMLMHead(config)
|
||||
self.sop_classifier = AlbertSOPHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.predictions.decoder
|
||||
@@ -876,7 +871,7 @@ class AlbertForMaskedLM(AlbertPreTrainedModel):
|
||||
self.albert = AlbertModel(config, add_pooling_layer=False)
|
||||
self.predictions = AlbertMLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.predictions.decoder
|
||||
@@ -970,7 +965,7 @@ class AlbertForSequenceClassification(AlbertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.classifier_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, self.config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ALBERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1058,7 +1053,7 @@ class AlbertForTokenClassification(AlbertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, self.config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ALBERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1146,7 +1141,7 @@ class AlbertForQuestionAnswering(AlbertPreTrainedModel):
|
||||
self.albert = AlbertModel(config, add_pooling_layer=False)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ALBERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1245,7 +1240,7 @@ class AlbertForMultipleChoice(AlbertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ALBERT_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -842,7 +842,7 @@ class BartModel(PretrainedBartModel):
|
||||
self.encoder = BartEncoder(config, self.shared)
|
||||
self.decoder = BartDecoder(config, self.shared)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BART_INPUTS_DOCSTRING)
|
||||
@add_code_sample_docstrings(
|
||||
@@ -875,10 +875,8 @@ class BartModel(PretrainedBartModel):
|
||||
if decoder_input_ids is None:
|
||||
use_cache = False
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
|
||||
@@ -455,14 +455,32 @@ class BertEncoder(nn.Module):
|
||||
all_hidden_states = all_hidden_states + (hidden_states,)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
|
||||
if getattr(self.config, "gradient_checkpointing", False):
|
||||
|
||||
def create_custom_forward(module):
|
||||
def custom_forward(*inputs):
|
||||
return module(*inputs, output_attentions)
|
||||
|
||||
return custom_forward
|
||||
|
||||
layer_outputs = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(layer_module),
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
)
|
||||
else:
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
hidden_states = layer_outputs[0]
|
||||
if output_attentions:
|
||||
all_attentions = all_attentions + (layer_outputs[1],)
|
||||
@@ -714,7 +732,7 @@ class BertModel(BertPreTrainedModel):
|
||||
|
||||
self.pooler = BertPooler(config) if add_pooling_layer else None
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -722,9 +740,6 @@ class BertModel(BertPreTrainedModel):
|
||||
def set_input_embeddings(self, value):
|
||||
self.embeddings.word_embeddings = value
|
||||
|
||||
def get_layers(self):
|
||||
return self.encoder.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -765,13 +780,11 @@ class BertModel(BertPreTrainedModel):
|
||||
- 1 for tokens that are **not masked**,
|
||||
- 0 for tokens that are **masked**.
|
||||
"""
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -852,7 +865,7 @@ class BertForPreTraining(BertPreTrainedModel):
|
||||
self.bert = BertModel(config)
|
||||
self.cls = BertPreTrainingHeads(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.cls.predictions.decoder
|
||||
@@ -965,7 +978,7 @@ class BertLMHeadModel(BertPreTrainedModel):
|
||||
self.bert = BertModel(config, add_pooling_layer=False)
|
||||
self.cls = BertOnlyMLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.cls.predictions.decoder
|
||||
@@ -1085,7 +1098,7 @@ class BertForMaskedLM(BertPreTrainedModel):
|
||||
self.bert = BertModel(config, add_pooling_layer=False)
|
||||
self.cls = BertOnlyMLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.cls.predictions.decoder
|
||||
@@ -1191,7 +1204,7 @@ class BertForNextSentencePrediction(BertPreTrainedModel):
|
||||
self.bert = BertModel(config)
|
||||
self.cls = BertOnlyNSPHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@replace_return_docstrings(output_type=NextSentencePredictorOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -1285,7 +1298,7 @@ class BertForSequenceClassification(BertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1369,7 +1382,7 @@ class BertForMultipleChoice(BertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BERT_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1464,7 +1477,7 @@ class BertForTokenClassification(BertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1554,7 +1567,7 @@ class BertForQuestionAnswering(BertPreTrainedModel):
|
||||
self.bert = BertModel(config, add_pooling_layer=False)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(BERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -277,7 +277,7 @@ class BertGenerationEncoder(BertGenerationPreTrainedModel):
|
||||
self.embeddings = BertGenerationEmbeddings(config)
|
||||
self.encoder = BertEncoder(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -285,9 +285,6 @@ class BertGenerationEncoder(BertGenerationPreTrainedModel):
|
||||
def set_input_embeddings(self, value):
|
||||
self.embeddings.word_embeddings = value
|
||||
|
||||
def get_layers(self):
|
||||
return self.encoder.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -325,13 +322,11 @@ class BertGenerationEncoder(BertGenerationPreTrainedModel):
|
||||
the cross-attention if the model is configured as a decoder. Mask values selected in ``[0, 1]``: ``1`` for
|
||||
tokens that are NOT MASKED, ``0`` for MASKED tokens.
|
||||
"""
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -421,7 +416,7 @@ class BertGenerationDecoder(BertGenerationPreTrainedModel):
|
||||
self.bert = BertGenerationEncoder(config)
|
||||
self.lm_head = BertGenerationOnlyLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head.decoder
|
||||
|
||||
@@ -149,7 +149,7 @@ class MultiHeadAttention(torch.nn.Module):
|
||||
k = torch.cat((past_key, k), dim=-2)
|
||||
v = torch.cat((past_value, v), dim=-2)
|
||||
|
||||
if use_cache:
|
||||
if use_cache is True:
|
||||
present = torch.stack((k, v))
|
||||
else:
|
||||
present = (None,)
|
||||
@@ -334,7 +334,7 @@ class CTRLModel(CTRLPreTrainedModel):
|
||||
)
|
||||
self.layernorm = nn.LayerNorm(config.n_embd, eps=config.layer_norm_epsilon)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.w
|
||||
@@ -342,9 +342,6 @@ class CTRLModel(CTRLPreTrainedModel):
|
||||
def set_input_embeddings(self, new_embeddings):
|
||||
self.w = new_embeddings
|
||||
|
||||
def get_layers(self):
|
||||
return self.h
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer}
|
||||
@@ -382,11 +379,9 @@ class CTRLModel(CTRLPreTrainedModel):
|
||||
past_key_values = kwargs.pop("past")
|
||||
assert kwargs == {}, f"Unexpected keyword arguments: {list(kwargs.keys())}."
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
use_cache = torch.tensor(use_cache if use_cache is not None else self.config.use_cache)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -464,18 +459,17 @@ class CTRLModel(CTRLPreTrainedModel):
|
||||
for i, (h, layer_past) in enumerate(zip(self.h, past_key_values)):
|
||||
if output_hidden_states:
|
||||
all_hidden_states = all_hidden_states + (hidden_states.view(*output_shape),)
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
outputs = h(
|
||||
hidden_states,
|
||||
mask,
|
||||
layer_past,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
use_cache,
|
||||
output_attentions,
|
||||
layer_past=layer_past,
|
||||
attention_mask=attention_mask,
|
||||
head_mask=head_mask[i],
|
||||
use_cache=use_cache,
|
||||
output_attentions=output_attentions,
|
||||
)
|
||||
hidden_states, present = outputs[:2]
|
||||
if use_cache:
|
||||
if use_cache is True:
|
||||
presents = presents + (present,)
|
||||
|
||||
if output_attentions:
|
||||
@@ -515,7 +509,7 @@ class CTRLLMHeadModel(CTRLPreTrainedModel):
|
||||
self.transformer = CTRLModel(config)
|
||||
self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=True)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
|
||||
@@ -406,9 +406,9 @@ class DebertaEncoder(nn.Module):
|
||||
next_kv,
|
||||
attention_mask,
|
||||
output_attentions,
|
||||
query_states,
|
||||
relative_pos,
|
||||
rel_embeddings,
|
||||
query_states=query_states,
|
||||
relative_pos=relative_pos,
|
||||
rel_embeddings=rel_embeddings,
|
||||
)
|
||||
if output_attentions:
|
||||
hidden_states, att_m = hidden_states
|
||||
@@ -843,7 +843,7 @@ class DebertaModel(DebertaPreTrainedModel):
|
||||
self.encoder = DebertaEncoder(config)
|
||||
self.z_steps = 0
|
||||
self.config = config
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -851,9 +851,6 @@ class DebertaModel(DebertaPreTrainedModel):
|
||||
def set_input_embeddings(self, new_embeddings):
|
||||
self.embeddings.word_embeddings = new_embeddings
|
||||
|
||||
def get_layers(self):
|
||||
return self.encoder.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -879,13 +876,11 @@ class DebertaModel(DebertaPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -973,7 +968,7 @@ class DebertaForSequenceClassification(DebertaPreTrainedModel):
|
||||
drop_out = self.config.hidden_dropout_prob if drop_out is None else drop_out
|
||||
self.dropout = StableDropout(drop_out)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.deberta.get_input_embeddings()
|
||||
|
||||
@@ -306,8 +306,9 @@ class Transformer(nn.Module):
|
||||
if output_hidden_states:
|
||||
all_hidden_states = all_hidden_states + (hidden_state,)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
layer_outputs = layer_module(hidden_state, attn_mask, layer_head_mask, output_attentions)
|
||||
layer_outputs = layer_module(
|
||||
x=hidden_state, attn_mask=attn_mask, head_mask=head_mask[i], output_attentions=output_attentions
|
||||
)
|
||||
hidden_state = layer_outputs[-1]
|
||||
|
||||
if output_attentions:
|
||||
@@ -419,7 +420,7 @@ class DistilBertModel(DistilBertPreTrainedModel):
|
||||
self.embeddings = Embeddings(config) # Embeddings
|
||||
self.transformer = Transformer(config) # Encoder
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -427,9 +428,6 @@ class DistilBertModel(DistilBertPreTrainedModel):
|
||||
def set_input_embeddings(self, new_embeddings):
|
||||
self.embeddings.word_embeddings = new_embeddings
|
||||
|
||||
def get_layers(self):
|
||||
return self.transformer.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -456,13 +454,11 @@ class DistilBertModel(DistilBertPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -506,7 +502,7 @@ class DistilBertForMaskedLM(DistilBertPreTrainedModel):
|
||||
self.vocab_layer_norm = nn.LayerNorm(config.dim, eps=1e-12)
|
||||
self.vocab_projector = nn.Linear(config.dim, config.vocab_size)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
self.mlm_loss_fct = nn.CrossEntropyLoss()
|
||||
|
||||
@@ -597,7 +593,7 @@ class DistilBertForSequenceClassification(DistilBertPreTrainedModel):
|
||||
self.classifier = nn.Linear(config.dim, config.num_labels)
|
||||
self.dropout = nn.Dropout(config.seq_classif_dropout)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(DISTILBERT_INPUTS_DOCSTRING.format("batch_size, num_choices"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -678,7 +674,7 @@ class DistilBertForQuestionAnswering(DistilBertPreTrainedModel):
|
||||
assert config.num_labels == 2
|
||||
self.dropout = nn.Dropout(config.qa_dropout)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(DISTILBERT_INPUTS_DOCSTRING.format("batch_size, num_choices"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -774,7 +770,7 @@ class DistilBertForTokenClassification(DistilBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.dropout)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(DISTILBERT_INPUTS_DOCSTRING)
|
||||
@add_code_sample_docstrings(
|
||||
@@ -858,7 +854,7 @@ class DistilBertForMultipleChoice(DistilBertPreTrainedModel):
|
||||
self.classifier = nn.Linear(config.dim, 1)
|
||||
self.dropout = nn.Dropout(config.seq_classif_dropout)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(
|
||||
DISTILBERT_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length")
|
||||
|
||||
@@ -157,7 +157,7 @@ class DPREncoder(PreTrainedModel):
|
||||
self.projection_dim = config.projection_dim
|
||||
if self.projection_dim > 0:
|
||||
self.encode_proj = nn.Linear(self.bert_model.config.hidden_size, config.projection_dim)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def forward(
|
||||
self,
|
||||
@@ -199,8 +199,8 @@ class DPREncoder(PreTrainedModel):
|
||||
return self.encode_proj.out_features
|
||||
return self.bert_model.config.hidden_size
|
||||
|
||||
def init_weights_and_layers(self):
|
||||
self.bert_model.init_weights_and_layers()
|
||||
def init_weights(self):
|
||||
self.bert_model.init_weights()
|
||||
if self.projection_dim > 0:
|
||||
self.encode_proj.apply(self.bert_model._init_weights)
|
||||
|
||||
@@ -214,7 +214,7 @@ class DPRSpanPredictor(PreTrainedModel):
|
||||
self.encoder = DPREncoder(config)
|
||||
self.qa_outputs = nn.Linear(self.encoder.embeddings_size, 2)
|
||||
self.qa_classifier = nn.Linear(self.encoder.embeddings_size, 1)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def forward(
|
||||
self,
|
||||
@@ -261,8 +261,8 @@ class DPRSpanPredictor(PreTrainedModel):
|
||||
attentions=outputs.attentions,
|
||||
)
|
||||
|
||||
def init_weights_and_layers(self):
|
||||
self.encoder.init_weights_and_layers()
|
||||
def init_weights(self):
|
||||
self.encoder.init_weights()
|
||||
|
||||
|
||||
##################
|
||||
@@ -281,8 +281,8 @@ class DPRPretrainedContextEncoder(PreTrainedModel):
|
||||
base_model_prefix = "ctx_encoder"
|
||||
authorized_missing_keys = [r"position_ids"]
|
||||
|
||||
def init_weights_and_layers(self):
|
||||
self.ctx_encoder.init_weights_and_layers()
|
||||
def init_weights(self):
|
||||
self.ctx_encoder.init_weights()
|
||||
|
||||
|
||||
class DPRPretrainedQuestionEncoder(PreTrainedModel):
|
||||
@@ -296,8 +296,8 @@ class DPRPretrainedQuestionEncoder(PreTrainedModel):
|
||||
base_model_prefix = "question_encoder"
|
||||
authorized_missing_keys = [r"position_ids"]
|
||||
|
||||
def init_weights_and_layers(self):
|
||||
self.question_encoder.init_weights_and_layers()
|
||||
def init_weights(self):
|
||||
self.question_encoder.init_weights()
|
||||
|
||||
|
||||
class DPRPretrainedReader(PreTrainedModel):
|
||||
@@ -311,8 +311,8 @@ class DPRPretrainedReader(PreTrainedModel):
|
||||
base_model_prefix = "span_predictor"
|
||||
authorized_missing_keys = [r"position_ids"]
|
||||
|
||||
def init_weights_and_layers(self):
|
||||
self.span_predictor.encoder.init_weights_and_layers()
|
||||
def init_weights(self):
|
||||
self.span_predictor.encoder.init_weights()
|
||||
self.span_predictor.qa_classifier.apply(self.span_predictor.encoder.bert_model._init_weights)
|
||||
self.span_predictor.qa_outputs.apply(self.span_predictor.encoder.bert_model._init_weights)
|
||||
|
||||
@@ -434,7 +434,7 @@ class DPRContextEncoder(DPRPretrainedContextEncoder):
|
||||
super().__init__(config)
|
||||
self.config = config
|
||||
self.ctx_encoder = DPREncoder(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(DPR_ENCODERS_INPUTS_DOCSTRING)
|
||||
@replace_return_docstrings(output_type=DPRContextEncoderOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -460,10 +460,8 @@ class DPRContextEncoder(DPRPretrainedContextEncoder):
|
||||
>>> embeddings = model(input_ids).pooler_output
|
||||
"""
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -514,7 +512,7 @@ class DPRQuestionEncoder(DPRPretrainedQuestionEncoder):
|
||||
super().__init__(config)
|
||||
self.config = config
|
||||
self.question_encoder = DPREncoder(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(DPR_ENCODERS_INPUTS_DOCSTRING)
|
||||
@replace_return_docstrings(output_type=DPRQuestionEncoderOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -539,10 +537,8 @@ class DPRQuestionEncoder(DPRPretrainedQuestionEncoder):
|
||||
>>> input_ids = tokenizer("Hello, is my dog cute ?", return_tensors='pt')["input_ids"]
|
||||
>>> embeddings = model(input_ids).pooler_output
|
||||
"""
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -593,7 +589,7 @@ class DPRReader(DPRPretrainedReader):
|
||||
super().__init__(config)
|
||||
self.config = config
|
||||
self.span_predictor = DPRSpanPredictor(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(DPR_READER_INPUTS_DOCSTRING)
|
||||
@replace_return_docstrings(output_type=DPRReaderOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -626,10 +622,8 @@ class DPRReader(DPRPretrainedReader):
|
||||
>>> relevance_logits = outputs.relevance_logits
|
||||
|
||||
"""
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
@@ -451,14 +451,32 @@ class ElectraEncoder(nn.Module):
|
||||
all_hidden_states = all_hidden_states + (hidden_states,)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
|
||||
if getattr(self.config, "gradient_checkpointing", False):
|
||||
|
||||
def create_custom_forward(module):
|
||||
def custom_forward(*inputs):
|
||||
return module(*inputs, output_attentions)
|
||||
|
||||
return custom_forward
|
||||
|
||||
layer_outputs = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(layer_module),
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
)
|
||||
else:
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
hidden_states = layer_outputs[0]
|
||||
if output_attentions:
|
||||
all_attentions = all_attentions + (layer_outputs[1],)
|
||||
@@ -659,7 +677,7 @@ class ElectraModel(ElectraPreTrainedModel):
|
||||
|
||||
self.encoder = ElectraEncoder(config)
|
||||
self.config = config
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -667,9 +685,6 @@ class ElectraModel(ElectraPreTrainedModel):
|
||||
def set_input_embeddings(self, value):
|
||||
self.embeddings.word_embeddings = value
|
||||
|
||||
def get_layers(self):
|
||||
return self.encoder.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -697,13 +712,11 @@ class ElectraModel(ElectraPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -776,7 +789,7 @@ class ElectraForSequenceClassification(ElectraPreTrainedModel):
|
||||
self.electra = ElectraModel(config)
|
||||
self.classifier = ElectraClassificationHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ELECTRA_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -858,7 +871,7 @@ class ElectraForPreTraining(ElectraPreTrainedModel):
|
||||
|
||||
self.electra = ElectraModel(config)
|
||||
self.discriminator_predictions = ElectraDiscriminatorPredictions(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ELECTRA_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@replace_return_docstrings(output_type=ElectraForPreTrainingOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -953,7 +966,7 @@ class ElectraForMaskedLM(ElectraPreTrainedModel):
|
||||
self.generator_predictions = ElectraGeneratorPredictions(config)
|
||||
|
||||
self.generator_lm_head = nn.Linear(config.embedding_size, config.vocab_size)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.generator_lm_head
|
||||
@@ -1045,7 +1058,7 @@ class ElectraForTokenClassification(ElectraPreTrainedModel):
|
||||
self.electra = ElectraModel(config)
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ELECTRA_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1132,7 +1145,7 @@ class ElectraForQuestionAnswering(ElectraPreTrainedModel):
|
||||
self.electra = ElectraModel(config)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ELECTRA_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1233,7 +1246,7 @@ class ElectraForMultipleChoice(ElectraPreTrainedModel):
|
||||
self.sequence_summary = SequenceSummary(config)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ELECTRA_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -162,10 +162,8 @@ class FlaubertModel(XLMModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -326,7 +324,7 @@ class FlaubertWithLMHeadModel(XLMWithLMHeadModel):
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
@@ -347,7 +345,7 @@ class FlaubertForSequenceClassification(XLMForSequenceClassification):
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
@@ -368,7 +366,7 @@ class FlaubertForTokenClassification(XLMForTokenClassification):
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
@@ -389,7 +387,7 @@ class FlaubertForQuestionAnsweringSimple(XLMForQuestionAnsweringSimple):
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
@@ -410,7 +408,7 @@ class FlaubertForQuestionAnswering(XLMForQuestionAnswering):
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
@@ -431,4 +429,4 @@ class FlaubertForMultipleChoice(XLMForMultipleChoice):
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
@@ -897,7 +897,7 @@ class FSMTModel(PretrainedFSMTModel):
|
||||
self.encoder = FSMTEncoder(config, encoder_embed_tokens)
|
||||
self.decoder = FSMTDecoder(config, decoder_embed_tokens)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(FSMT_INPUTS_DOCSTRING)
|
||||
@add_code_sample_docstrings(
|
||||
@@ -930,10 +930,8 @@ class FSMTModel(PretrainedFSMTModel):
|
||||
if decoder_input_ids is None:
|
||||
use_cache = False
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
|
||||
@@ -648,7 +648,7 @@ class FunnelEncoder(nn.Module):
|
||||
key = value = hidden if self.config.pool_q_only else pooled_hidden
|
||||
else:
|
||||
query = key = value = hidden
|
||||
layer_output = layer(query, key, value, attention_inputs, output_attentions)
|
||||
layer_output = layer(query, key, value, attention_inputs, output_attentions=output_attentions)
|
||||
hidden = layer_output[0]
|
||||
if do_pooling:
|
||||
attention_inputs = self.attention_structure.post_attention_pooling(attention_inputs)
|
||||
@@ -898,7 +898,7 @@ class FunnelBaseModel(FunnelPreTrainedModel):
|
||||
self.embeddings = FunnelEmbeddings(config)
|
||||
self.encoder = FunnelEncoder(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -906,9 +906,6 @@ class FunnelBaseModel(FunnelPreTrainedModel):
|
||||
def set_input_embeddings(self, new_embeddings):
|
||||
self.embeddings.word_embeddings = new_embeddings
|
||||
|
||||
def get_layers(self):
|
||||
return [layer for block in self.encoder.blocks for layer in block]
|
||||
|
||||
@add_start_docstrings_to_model_forward(FUNNEL_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
tokenizer_class=_TOKENIZER_FOR_DOC,
|
||||
@@ -928,10 +925,8 @@ class FunnelBaseModel(FunnelPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -980,7 +975,7 @@ class FunnelModel(FunnelPreTrainedModel):
|
||||
self.encoder = FunnelEncoder(config)
|
||||
self.decoder = FunnelDecoder(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -1005,9 +1000,7 @@ class FunnelModel(FunnelPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
@@ -1087,7 +1080,7 @@ class FunnelForPreTraining(FunnelPreTrainedModel):
|
||||
|
||||
self.funnel = FunnelModel(config)
|
||||
self.discriminator_predictions = FunnelDiscriminatorPredictions(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(FUNNEL_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@replace_return_docstrings(output_type=FunnelForPreTrainingOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -1169,7 +1162,7 @@ class FunnelForMaskedLM(FunnelPreTrainedModel):
|
||||
self.funnel = FunnelModel(config)
|
||||
self.lm_head = nn.Linear(config.d_model, config.vocab_size)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
@@ -1245,7 +1238,7 @@ class FunnelForSequenceClassification(FunnelPreTrainedModel):
|
||||
|
||||
self.funnel = FunnelBaseModel(config)
|
||||
self.classifier = FunnelClassificationHead(config, config.num_labels)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(FUNNEL_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1322,7 +1315,7 @@ class FunnelForMultipleChoice(FunnelPreTrainedModel):
|
||||
|
||||
self.funnel = FunnelBaseModel(config)
|
||||
self.classifier = FunnelClassificationHead(config, 1)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(FUNNEL_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1408,7 +1401,7 @@ class FunnelForTokenClassification(FunnelPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(FUNNEL_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1490,7 +1483,7 @@ class FunnelForQuestionAnswering(FunnelPreTrainedModel):
|
||||
self.funnel = FunnelModel(config)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(FUNNEL_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -227,7 +227,7 @@ class Attention(nn.Module):
|
||||
key = torch.cat((past_key, key), dim=-1)
|
||||
value = torch.cat((past_value, value), dim=-2)
|
||||
|
||||
if use_cache:
|
||||
if use_cache is True:
|
||||
present = torch.stack((key.transpose(-2, -1), value)) # transpose to have same shapes for stacking
|
||||
else:
|
||||
present = (None,)
|
||||
@@ -317,7 +317,7 @@ class Block(nn.Module):
|
||||
# residual connection
|
||||
hidden_states = hidden_states + feed_forward_hidden_states
|
||||
|
||||
outputs = (hidden_states,) + tuple(outputs)
|
||||
outputs = [hidden_states] + outputs
|
||||
return outputs # hidden_states, present, (cross_attentions, attentions)
|
||||
|
||||
|
||||
@@ -487,7 +487,7 @@ class GPT2Model(GPT2PreTrainedModel):
|
||||
self.h = nn.ModuleList([Block(config.n_ctx, config, scale=True) for _ in range(config.n_layer)])
|
||||
self.ln_f = nn.LayerNorm(config.n_embd, eps=config.layer_norm_epsilon)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.wte
|
||||
@@ -495,9 +495,6 @@ class GPT2Model(GPT2PreTrainedModel):
|
||||
def set_input_embeddings(self, new_embeddings):
|
||||
self.wte = new_embeddings
|
||||
|
||||
def get_layers(self):
|
||||
return self.h
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer}
|
||||
@@ -537,14 +534,12 @@ class GPT2Model(GPT2PreTrainedModel):
|
||||
past_key_values = kwargs.pop("past")
|
||||
assert kwargs == {}, f"Unexpected keyword arguments: {list(kwargs.keys())}."
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
use_cache = torch.tensor(use_cache if use_cache is not None else self.config.use_cache)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -629,20 +624,38 @@ class GPT2Model(GPT2PreTrainedModel):
|
||||
if output_hidden_states:
|
||||
all_hidden_states = all_hidden_states + (hidden_states.view(*output_shape),)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
outputs = block(
|
||||
hidden_states,
|
||||
layer_past,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
use_cache,
|
||||
output_attentions,
|
||||
)
|
||||
if getattr(self.config, "gradient_checkpointing", False):
|
||||
|
||||
def create_custom_forward(module):
|
||||
def custom_forward(*inputs):
|
||||
# checkpointing only works with tuple returns, not with lists
|
||||
return tuple(output for output in module(*inputs, use_cache, output_attentions))
|
||||
|
||||
return custom_forward
|
||||
|
||||
outputs = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(block),
|
||||
hidden_states,
|
||||
layer_past,
|
||||
attention_mask,
|
||||
head_mask[i],
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
)
|
||||
else:
|
||||
outputs = block(
|
||||
hidden_states,
|
||||
layer_past=layer_past,
|
||||
attention_mask=attention_mask,
|
||||
head_mask=head_mask[i],
|
||||
encoder_hidden_states=encoder_hidden_states,
|
||||
encoder_attention_mask=encoder_attention_mask,
|
||||
use_cache=use_cache,
|
||||
output_attentions=output_attentions,
|
||||
)
|
||||
|
||||
hidden_states, present = outputs[:2]
|
||||
if use_cache:
|
||||
if use_cache is True:
|
||||
presents = presents + (present,)
|
||||
|
||||
if output_attentions:
|
||||
@@ -651,7 +664,6 @@ class GPT2Model(GPT2PreTrainedModel):
|
||||
hidden_states = self.ln_f(hidden_states)
|
||||
|
||||
hidden_states = hidden_states.view(*output_shape)
|
||||
|
||||
# Add last hidden state
|
||||
if output_hidden_states:
|
||||
all_hidden_states = all_hidden_states + (hidden_states,)
|
||||
@@ -682,7 +694,7 @@ class GPT2LMHeadModel(GPT2PreTrainedModel):
|
||||
self.transformer = GPT2Model(config)
|
||||
self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=False)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
@@ -809,7 +821,7 @@ class GPT2DoubleHeadsModel(GPT2PreTrainedModel):
|
||||
self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=False)
|
||||
self.multiple_choice_head = SequenceSummary(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
@@ -973,7 +985,7 @@ class GPT2ForSequenceClassification(GPT2PreTrainedModel):
|
||||
self.transformer = GPT2Model(config)
|
||||
self.score = nn.Linear(config.n_embd, self.num_labels, bias=False)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(GPT2_INPUTS_DOCSTRING)
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -380,14 +380,32 @@ class LayoutLMEncoder(nn.Module):
|
||||
all_hidden_states = all_hidden_states + (hidden_states,)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
|
||||
if getattr(self.config, "gradient_checkpointing", False):
|
||||
|
||||
def create_custom_forward(module):
|
||||
def custom_forward(*inputs):
|
||||
return module(*inputs, output_attentions)
|
||||
|
||||
return custom_forward
|
||||
|
||||
layer_outputs = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(layer_module),
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
)
|
||||
else:
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
hidden_states = layer_outputs[0]
|
||||
if output_attentions:
|
||||
all_attentions = all_attentions + (layer_outputs[1],)
|
||||
@@ -573,7 +591,7 @@ class LayoutLMModel(LayoutLMPreTrainedModel):
|
||||
self.encoder = LayoutLMEncoder(config)
|
||||
self.pooler = LayoutLMPooler(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -581,9 +599,6 @@ class LayoutLMModel(LayoutLMPreTrainedModel):
|
||||
def set_input_embeddings(self, value):
|
||||
self.embeddings.word_embeddings = value
|
||||
|
||||
def get_layers(self):
|
||||
return self.encoder.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -640,13 +655,11 @@ class LayoutLMModel(LayoutLMPreTrainedModel):
|
||||
return_dict (bool, optional):
|
||||
If set to True, the model will return a ModelOutput instead of a plain tuple.
|
||||
"""
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -723,7 +736,7 @@ class LayoutLMForMaskedLM(LayoutLMPreTrainedModel):
|
||||
self.layoutlm = LayoutLMModel(config)
|
||||
self.cls = LayoutLMOnlyMLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.layoutlm.embeddings.word_embeddings
|
||||
@@ -814,7 +827,7 @@ class LayoutLMForTokenClassification(LayoutLMPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.layoutlm.embeddings.word_embeddings
|
||||
|
||||
@@ -1106,7 +1106,7 @@ class LongformerModel(LongformerPreTrainedModel):
|
||||
self.encoder = LongformerEncoder(config)
|
||||
self.pooler = LongformerPooler(config) if add_pooling_layer else None
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -1223,10 +1223,8 @@ class LongformerModel(LongformerPreTrainedModel):
|
||||
>>> pooled_output = outputs.pooler_output
|
||||
"""
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -1305,7 +1303,7 @@ class LongformerForMaskedLM(LongformerPreTrainedModel):
|
||||
self.longformer = LongformerModel(config, add_pooling_layer=False)
|
||||
self.lm_head = LongformerLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head.decoder
|
||||
@@ -1412,7 +1410,7 @@ class LongformerForSequenceClassification(LongformerPreTrainedModel):
|
||||
self.longformer = LongformerModel(config, add_pooling_layer=False)
|
||||
self.classifier = LongformerClassificationHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(LONGFORMER_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1521,7 +1519,7 @@ class LongformerForQuestionAnswering(LongformerPreTrainedModel):
|
||||
self.longformer = LongformerModel(config, add_pooling_layer=False)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(LONGFORMER_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@replace_return_docstrings(output_type=QuestionAnsweringModelOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -1655,7 +1653,7 @@ class LongformerForTokenClassification(LongformerPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(LONGFORMER_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1742,7 +1740,7 @@ class LongformerForMultipleChoice(LongformerPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(
|
||||
LONGFORMER_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length")
|
||||
|
||||
@@ -885,7 +885,7 @@ class LxmertModel(LxmertPreTrainedModel):
|
||||
self.embeddings = LxmertEmbeddings(config)
|
||||
self.encoder = LxmertEncoder(config)
|
||||
self.pooler = LxmertPooler(config)
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -914,10 +914,8 @@ class LxmertModel(LxmertPreTrainedModel):
|
||||
return_dict=None,
|
||||
):
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -1042,7 +1040,7 @@ class LxmertForPreTraining(LxmertPreTrainedModel):
|
||||
self.answer_head = LxmertVisualAnswerHead(config, self.num_qa_labels)
|
||||
|
||||
# Weight initialization
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
# Loss functions
|
||||
self.loss_fcts = {
|
||||
@@ -1287,7 +1285,7 @@ class LxmertForQuestionAnswering(LxmertPreTrainedModel):
|
||||
self.answer_head = LxmertVisualAnswerHead(config, self.num_qa_labels)
|
||||
|
||||
# Weight initialization
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
# Loss function
|
||||
self.loss = CrossEntropyLoss()
|
||||
|
||||
@@ -218,10 +218,8 @@ class MMBTModel(nn.Module, ModuleUtilsMixin):
|
||||
encoder = ImageEncoder(args)
|
||||
mmbt = MMBTModel(config, transformer, encoder)
|
||||
"""
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
@@ -566,11 +566,10 @@ class MobileBertEncoder(nn.Module):
|
||||
if output_hidden_states:
|
||||
all_hidden_states = all_hidden_states + (hidden_states,)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
head_mask[i],
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
@@ -822,7 +821,7 @@ class MobileBertModel(MobileBertPreTrainedModel):
|
||||
|
||||
self.pooler = MobileBertPooler(config) if add_pooling_layer else None
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -830,9 +829,6 @@ class MobileBertModel(MobileBertPreTrainedModel):
|
||||
def set_input_embeddings(self, value):
|
||||
self.embeddings.word_embeddings = value
|
||||
|
||||
def get_layers(self):
|
||||
return self.encoder.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -862,13 +858,11 @@ class MobileBertModel(MobileBertPreTrainedModel):
|
||||
output_attentions=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -950,7 +944,7 @@ class MobileBertForPreTraining(MobileBertPreTrainedModel):
|
||||
self.mobilebert = MobileBertModel(config)
|
||||
self.cls = MobileBertPreTrainingHeads(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.cls.predictions.decoder
|
||||
@@ -1068,7 +1062,7 @@ class MobileBertForMaskedLM(MobileBertPreTrainedModel):
|
||||
self.cls = MobileBertOnlyMLMHead(config)
|
||||
self.config = config
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.cls.predictions.decoder
|
||||
@@ -1188,7 +1182,7 @@ class MobileBertForNextSentencePrediction(MobileBertPreTrainedModel):
|
||||
self.mobilebert = MobileBertModel(config)
|
||||
self.cls = MobileBertOnlyNSPHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MOBILEBERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@replace_return_docstrings(output_type=NextSentencePredictorOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -1280,7 +1274,7 @@ class MobileBertForSequenceClassification(MobileBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, self.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MOBILEBERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1365,7 +1359,7 @@ class MobileBertForQuestionAnswering(MobileBertPreTrainedModel):
|
||||
self.mobilebert = MobileBertModel(config, add_pooling_layer=False)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MOBILEBERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1464,7 +1458,7 @@ class MobileBertForMultipleChoice(MobileBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(
|
||||
MOBILEBERT_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length")
|
||||
@@ -1561,7 +1555,7 @@ class MobileBertForTokenClassification(MobileBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(MOBILEBERT_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -268,7 +268,7 @@ class Block(nn.Module):
|
||||
h = self.ln_2(n + m)
|
||||
|
||||
outputs = [h] + attn_outputs[1:]
|
||||
return tuple(outputs)
|
||||
return outputs
|
||||
|
||||
|
||||
class OpenAIGPTPreTrainedModel(PreTrainedModel):
|
||||
@@ -412,7 +412,7 @@ class OpenAIGPTModel(OpenAIGPTPreTrainedModel):
|
||||
self.h = nn.ModuleList([Block(config.n_ctx, config, scale=True) for _ in range(config.n_layer)])
|
||||
|
||||
self.register_buffer("position_ids", torch.arange(config.n_positions))
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.tokens_embed
|
||||
@@ -420,9 +420,6 @@ class OpenAIGPTModel(OpenAIGPTPreTrainedModel):
|
||||
def set_input_embeddings(self, new_embeddings):
|
||||
self.tokens_embed = new_embeddings
|
||||
|
||||
def get_layers(self):
|
||||
return self.h
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer}
|
||||
@@ -449,13 +446,11 @@ class OpenAIGPTModel(OpenAIGPTPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -510,8 +505,7 @@ class OpenAIGPTModel(OpenAIGPTPreTrainedModel):
|
||||
if output_hidden_states:
|
||||
all_hidden_states = all_hidden_states + (hidden_states.view(*output_shape),)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
outputs = block(hidden_states, attention_mask, layer_head_mask, output_attentions)
|
||||
outputs = block(hidden_states, attention_mask, head_mask[i], output_attentions=output_attentions)
|
||||
hidden_states = outputs[0]
|
||||
if output_attentions:
|
||||
all_attentions = all_attentions + (outputs[1],)
|
||||
@@ -544,7 +538,7 @@ class OpenAIGPTLMHeadModel(OpenAIGPTPreTrainedModel):
|
||||
self.transformer = OpenAIGPTModel(config)
|
||||
self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=False)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
@@ -630,7 +624,7 @@ class OpenAIGPTDoubleHeadsModel(OpenAIGPTPreTrainedModel):
|
||||
self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=False)
|
||||
self.multiple_choice_head = SequenceSummary(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
@@ -758,7 +752,7 @@ class OpenAIGPTForSequenceClassification(OpenAIGPTPreTrainedModel):
|
||||
self.transformer = OpenAIGPTModel(config)
|
||||
self.score = nn.Linear(config.n_embd, self.num_labels, bias=False)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(OPENAI_GPT_INPUTS_DOCSTRING)
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -1135,7 +1135,7 @@ class ProphetNetEncoder(ProphetNetPreTrainedModel):
|
||||
|
||||
self.layers = nn.ModuleList([ProphetNetEncoderLayer(config) for _ in range(config.num_encoder_layers)])
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.word_embeddings
|
||||
@@ -1170,10 +1170,8 @@ class ProphetNetEncoder(ProphetNetPreTrainedModel):
|
||||
>>> last_hidden_states = outputs.last_hidden_state
|
||||
"""
|
||||
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -1253,7 +1251,7 @@ class ProphetNetDecoder(ProphetNetPreTrainedModel):
|
||||
self.layers = nn.ModuleList([ProphetNetDecoderLayer(config) for _ in range(config.num_decoder_layers)])
|
||||
self.embeddings_layer_norm = ProphetNetLayerNorm(config.hidden_size)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.word_embeddings
|
||||
@@ -1312,10 +1310,8 @@ class ProphetNetDecoder(ProphetNetPreTrainedModel):
|
||||
>>> last_hidden_states = outputs.last_hidden_state
|
||||
"""
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -1563,7 +1559,7 @@ class ProphetNetModel(ProphetNetPreTrainedModel):
|
||||
decoder_config.is_encoder_decoder = False
|
||||
self.decoder = ProphetNetDecoder(decoder_config, self.word_embeddings)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.word_embeddings
|
||||
@@ -1615,10 +1611,8 @@ class ProphetNetModel(ProphetNetPreTrainedModel):
|
||||
"""
|
||||
|
||||
use_cache == use_cache if use_cache is not None else self.config.use_cache
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -1677,7 +1671,7 @@ class ProphetNetForConditionalGeneration(ProphetNetPreTrainedModel):
|
||||
|
||||
self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
@@ -1865,7 +1859,7 @@ class ProphetNetForCausalLM(ProphetNetPreTrainedModel):
|
||||
|
||||
self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.decoder.word_embeddings
|
||||
|
||||
@@ -542,10 +542,8 @@ class RagModel(RagPreTrainedModel):
|
||||
"""
|
||||
n_docs = n_docs if n_docs is not None else self.config.n_docs
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
output_retrieved = output_retrieved if output_retrieved is not None else self.config.output_retrieved
|
||||
|
||||
@@ -1977,7 +1977,7 @@ class ReformerModel(ReformerPreTrainedModel):
|
||||
self.embeddings = ReformerEmbeddings(config)
|
||||
self.encoder = ReformerEncoder(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -2015,10 +2015,8 @@ class ReformerModel(ReformerPreTrainedModel):
|
||||
return_dict=None,
|
||||
):
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -2194,7 +2192,7 @@ class ReformerModelWithLMHead(ReformerPreTrainedModel):
|
||||
self.reformer = ReformerModel(config)
|
||||
self.lm_head = ReformerOnlyLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head.decoder
|
||||
@@ -2306,7 +2304,7 @@ class ReformerForMaskedLM(ReformerPreTrainedModel):
|
||||
self.reformer = ReformerModel(config)
|
||||
self.lm_head = ReformerOnlyLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head.decoder
|
||||
@@ -2389,7 +2387,7 @@ class ReformerForSequenceClassification(ReformerPreTrainedModel):
|
||||
if config.is_decoder is True:
|
||||
logger.warning("You might want to disable causal masking for sequence classification")
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(REFORMER_INPUTS_DOCSTRING)
|
||||
@add_code_sample_docstrings(
|
||||
@@ -2491,7 +2489,7 @@ class ReformerForQuestionAnswering(ReformerPreTrainedModel):
|
||||
# 2 * config.hidden_size because we use reversible residual layers
|
||||
self.qa_outputs = nn.Linear(2 * config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(REFORMER_INPUTS_DOCSTRING)
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -95,7 +95,7 @@ class RetriBertModel(RetriBertPreTrainedModel):
|
||||
|
||||
self.ce_loss = nn.CrossEntropyLoss(reduction="mean")
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def embed_sentences_checkpointed(
|
||||
self,
|
||||
|
||||
@@ -399,14 +399,32 @@ class RobertaEncoder(nn.Module):
|
||||
all_hidden_states = all_hidden_states + (hidden_states,)
|
||||
|
||||
layer_head_mask = head_mask[i] if head_mask is not None else None
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
|
||||
if getattr(self.config, "gradient_checkpointing", False):
|
||||
|
||||
def create_custom_forward(module):
|
||||
def custom_forward(*inputs):
|
||||
return module(*inputs, output_attentions)
|
||||
|
||||
return custom_forward
|
||||
|
||||
layer_outputs = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(layer_module),
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
)
|
||||
else:
|
||||
layer_outputs = layer_module(
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
layer_head_mask,
|
||||
encoder_hidden_states,
|
||||
encoder_attention_mask,
|
||||
output_attentions,
|
||||
)
|
||||
hidden_states = layer_outputs[0]
|
||||
if output_attentions:
|
||||
all_attentions = all_attentions + (layer_outputs[1],)
|
||||
@@ -561,7 +579,7 @@ class RobertaModel(RobertaPreTrainedModel):
|
||||
|
||||
self.pooler = RobertaPooler(config) if add_pooling_layer else None
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -569,9 +587,6 @@ class RobertaModel(RobertaPreTrainedModel):
|
||||
def set_input_embeddings(self, value):
|
||||
self.embeddings.word_embeddings = value
|
||||
|
||||
def get_layers(self):
|
||||
return self.encoder.layer
|
||||
|
||||
def _prune_heads(self, heads_to_prune):
|
||||
"""
|
||||
Prunes heads of the model. heads_to_prune: dict of {layer_num: list of heads to prune in this layer} See base
|
||||
@@ -611,13 +626,11 @@ class RobertaModel(RobertaPreTrainedModel):
|
||||
the cross-attention if the model is configured as a decoder. Mask values selected in ``[0, 1]``: ``1`` for
|
||||
tokens that are NOT MASKED, ``0`` for MASKED tokens.
|
||||
"""
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = torch.tensor(return_dict if return_dict is not None else self.config.use_return_dict)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
@@ -700,7 +713,7 @@ class RobertaForCausalLM(RobertaPreTrainedModel):
|
||||
self.roberta = RobertaModel(config, add_pooling_layer=False)
|
||||
self.lm_head = RobertaLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head.decoder
|
||||
@@ -820,7 +833,7 @@ class RobertaForMaskedLM(RobertaPreTrainedModel):
|
||||
self.roberta = RobertaModel(config, add_pooling_layer=False)
|
||||
self.lm_head = RobertaLMHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head.decoder
|
||||
@@ -941,7 +954,7 @@ class RobertaForSequenceClassification(RobertaPreTrainedModel):
|
||||
self.roberta = RobertaModel(config, add_pooling_layer=False)
|
||||
self.classifier = RobertaClassificationHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ROBERTA_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1024,7 +1037,7 @@ class RobertaForMultipleChoice(RobertaPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ROBERTA_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1118,7 +1131,7 @@ class RobertaForTokenClassification(RobertaPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ROBERTA_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1227,7 +1240,7 @@ class RobertaForQuestionAnswering(RobertaPreTrainedModel):
|
||||
self.roberta = RobertaModel(config, add_pooling_layer=False)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(ROBERTA_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -502,7 +502,7 @@ class SqueezeBertModel(SqueezeBertPreTrainedModel):
|
||||
self.encoder = SqueezeBertEncoder(config)
|
||||
self.pooler = SqueezeBertPooler(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embeddings.word_embeddings
|
||||
@@ -537,10 +537,8 @@ class SqueezeBertModel(SqueezeBertPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -602,7 +600,7 @@ class SqueezeBertForMaskedLM(SqueezeBertPreTrainedModel):
|
||||
self.transformer = SqueezeBertModel(config)
|
||||
self.lm_head = nn.Linear(config.embedding_size, config.vocab_size)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_head
|
||||
@@ -683,7 +681,7 @@ class SqueezeBertForSequenceClassification(SqueezeBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, self.config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(SQUEEZEBERT_INPUTS_DOCSTRING.format("(batch_size, sequence_length)"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -767,7 +765,7 @@ class SqueezeBertForMultipleChoice(SqueezeBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(
|
||||
SQUEEZEBERT_INPUTS_DOCSTRING.format("(batch_size, num_choices, sequence_length)")
|
||||
@@ -861,7 +859,7 @@ class SqueezeBertForTokenClassification(SqueezeBertPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(SQUEEZEBERT_INPUTS_DOCSTRING.format("(batch_size, sequence_length)"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -948,7 +946,7 @@ class SqueezeBertForQuestionAnswering(SqueezeBertPreTrainedModel):
|
||||
self.transformer = SqueezeBertModel(config)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(SQUEEZEBERT_INPUTS_DOCSTRING.format("(batch_size, sequence_length)"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -657,7 +657,7 @@ class T5Stack(T5PreTrainedModel):
|
||||
self.final_layer_norm = T5LayerNorm(config.d_model, eps=config.layer_norm_epsilon)
|
||||
self.dropout = nn.Dropout(config.dropout_rate)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.embed_tokens
|
||||
@@ -684,10 +684,8 @@ class T5Stack(T5PreTrainedModel):
|
||||
):
|
||||
|
||||
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -921,7 +919,7 @@ class T5Model(T5PreTrainedModel):
|
||||
decoder_config.num_layers = config.num_decoder_layers
|
||||
self.decoder = T5Stack(decoder_config, self.shared)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.shared
|
||||
@@ -1069,7 +1067,7 @@ class T5ForConditionalGeneration(T5PreTrainedModel):
|
||||
|
||||
self.lm_head = nn.Linear(config.d_model, config.vocab_size, bias=False)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.shared
|
||||
|
||||
@@ -785,7 +785,7 @@ class TransfoXLModel(TransfoXLPreTrainedModel):
|
||||
else: # learnable embeddings and absolute embeddings
|
||||
raise NotImplementedError # Removed these to avoid maintaining dead code - They are not used in our pretrained checkpoint
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.word_emb
|
||||
@@ -852,10 +852,8 @@ class TransfoXLModel(TransfoXLPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -989,7 +987,7 @@ class TransfoXLLMHeadModel(TransfoXLPreTrainedModel):
|
||||
config.vocab_size, config.d_embed, config.d_model, config.cutoffs, div_val=config.div_val
|
||||
)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def tie_weights(self):
|
||||
"""
|
||||
|
||||
@@ -14,7 +14,6 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import functools
|
||||
import inspect
|
||||
import os
|
||||
import re
|
||||
@@ -666,28 +665,9 @@ class PreTrainedModel(nn.Module, ModuleUtilsMixin, GenerationMixin):
|
||||
|
||||
return new_embeddings
|
||||
|
||||
def get_layers(self):
|
||||
"""
|
||||
Returns the model's transformer layers.
|
||||
|
||||
Returns:
|
||||
:obj:`List[nn.Module]` or :obj:`nn.ModuleList`: A list or :obj:`nn.ModuleList` containing all transformer
|
||||
layers.
|
||||
"""
|
||||
base_model = getattr(self, self.base_model_prefix, self)
|
||||
|
||||
if base_model is not self:
|
||||
return base_model.get_layers()
|
||||
else:
|
||||
raise NotImplementedError
|
||||
|
||||
def init_weights(self):
|
||||
# Backwards compatibility
|
||||
self.init_weights_and_layers()
|
||||
|
||||
def init_weights_and_layers(self):
|
||||
"""
|
||||
Initializes and prunes weights if needed. Sets up gradient checkpointing if in the configuration.
|
||||
Initializes and prunes weights if needed.
|
||||
"""
|
||||
# Initialize weights
|
||||
self.apply(self._init_weights)
|
||||
@@ -699,11 +679,6 @@ class PreTrainedModel(nn.Module, ModuleUtilsMixin, GenerationMixin):
|
||||
# Tie weights if needed
|
||||
self.tie_weights()
|
||||
|
||||
# Gradient checkpointing if needed
|
||||
if self.config.gradient_checkpointing:
|
||||
for layer in self.get_layers():
|
||||
layer.forward = functools.partial(torch.utils.checkpoint.checkpoint, layer.forward)
|
||||
|
||||
def prune_heads(self, heads_to_prune: Dict[int, List[int]]):
|
||||
"""
|
||||
Prunes heads of the base model.
|
||||
|
||||
@@ -469,7 +469,7 @@ class XLMModel(XLMPreTrainedModel):
|
||||
if self.attentions[int(layer)].n_heads == config.n_heads:
|
||||
self.prune_heads({int(layer): list(map(int, heads))})
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
self.register_buffer("position_ids", torch.arange(config.max_position_embeddings).expand((1, -1)))
|
||||
|
||||
def get_input_embeddings(self):
|
||||
@@ -508,10 +508,8 @@ class XLMModel(XLMPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -687,7 +685,7 @@ class XLMWithLMHeadModel(XLMPreTrainedModel):
|
||||
self.transformer = XLMModel(config)
|
||||
self.pred_layer = XLMPredLayer(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.pred_layer.proj
|
||||
@@ -781,7 +779,7 @@ class XLMForSequenceClassification(XLMPreTrainedModel):
|
||||
self.transformer = XLMModel(config)
|
||||
self.sequence_summary = SequenceSummary(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLM_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -868,7 +866,7 @@ class XLMForQuestionAnsweringSimple(XLMPreTrainedModel):
|
||||
self.transformer = XLMModel(config)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLM_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -972,7 +970,7 @@ class XLMForQuestionAnswering(XLMPreTrainedModel):
|
||||
self.transformer = XLMModel(config)
|
||||
self.qa_outputs = SQuADHead(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLM_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@replace_return_docstrings(output_type=XLMForQuestionAnsweringOutput, config_class=_CONFIG_FOR_DOC)
|
||||
@@ -1091,7 +1089,7 @@ class XLMForTokenClassification(XLMPreTrainedModel):
|
||||
self.dropout = nn.Dropout(config.dropout)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLM_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1184,7 +1182,7 @@ class XLMForMultipleChoice(XLMPreTrainedModel):
|
||||
self.sequence_summary = SequenceSummary(config)
|
||||
self.logits_proj = nn.Linear(config.num_labels, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLM_INPUTS_DOCSTRING.format("batch_size, num_choicec, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
|
||||
@@ -952,7 +952,7 @@ class XLNetModel(XLNetPreTrainedModel):
|
||||
self.layer = nn.ModuleList([XLNetLayer(config) for _ in range(config.n_layer)])
|
||||
self.dropout = nn.Dropout(config.dropout)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_input_embeddings(self):
|
||||
return self.word_embedding
|
||||
@@ -1087,10 +1087,8 @@ class XLNetModel(XLNetPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
@@ -1297,7 +1295,7 @@ class XLNetLMHeadModel(XLNetPreTrainedModel):
|
||||
self.transformer = XLNetModel(config)
|
||||
self.lm_loss = nn.Linear(config.d_model, config.vocab_size, bias=True)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
def get_output_embeddings(self):
|
||||
return self.lm_loss
|
||||
@@ -1465,7 +1463,7 @@ class XLNetForSequenceClassification(XLNetPreTrainedModel):
|
||||
self.sequence_summary = SequenceSummary(config)
|
||||
self.logits_proj = nn.Linear(config.d_model, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLNET_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1558,7 +1556,7 @@ class XLNetForTokenClassification(XLNetPreTrainedModel):
|
||||
self.transformer = XLNetModel(config)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLNET_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1655,7 +1653,7 @@ class XLNetForMultipleChoice(XLNetPreTrainedModel):
|
||||
self.sequence_summary = SequenceSummary(config)
|
||||
self.logits_proj = nn.Linear(config.d_model, 1)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLNET_INPUTS_DOCSTRING.format("batch_size, num_choices, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1756,7 +1754,7 @@ class XLNetForQuestionAnsweringSimple(XLNetPreTrainedModel):
|
||||
self.transformer = XLNetModel(config)
|
||||
self.qa_outputs = nn.Linear(config.hidden_size, config.num_labels)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLNET_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@add_code_sample_docstrings(
|
||||
@@ -1868,7 +1866,7 @@ class XLNetForQuestionAnswering(XLNetPreTrainedModel):
|
||||
self.end_logits = PoolerEndLogits(config)
|
||||
self.answer_class = PoolerAnswerClass(config)
|
||||
|
||||
self.init_weights_and_layers()
|
||||
self.init_weights()
|
||||
|
||||
@add_start_docstrings_to_model_forward(XLNET_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
|
||||
@replace_return_docstrings(output_type=XLNetForQuestionAnsweringOutput, config_class=_CONFIG_FOR_DOC)
|
||||
|
||||
@@ -26,11 +26,6 @@ class DataCollatorForLanguageModeling:
|
||||
requires_pytorch(self)
|
||||
|
||||
|
||||
class DataCollatorForNextSentencePrediction:
|
||||
def __init__(self, *args, **kwargs):
|
||||
requires_pytorch(self)
|
||||
|
||||
|
||||
class DataCollatorForPermutationLanguageModeling:
|
||||
def __init__(self, *args, **kwargs):
|
||||
requires_pytorch(self)
|
||||
|
||||
@@ -328,10 +328,8 @@ class XxxModel(XxxPreTrainedModel):
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
):
|
||||
output_attentions = torch.tensor(
|
||||
output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
)
|
||||
output_hidden_states = torch.tensor(
|
||||
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
||||
output_hidden_states = (
|
||||
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
||||
)
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
|
||||
@@ -12,9 +12,7 @@ if is_torch_available():
|
||||
|
||||
from transformers import (
|
||||
DataCollatorForLanguageModeling,
|
||||
DataCollatorForNextSentencePrediction,
|
||||
DataCollatorForPermutationLanguageModeling,
|
||||
DataCollatorForSOP,
|
||||
DataCollatorForTokenClassification,
|
||||
DataCollatorWithPadding,
|
||||
default_data_collator,
|
||||
@@ -201,13 +199,16 @@ class DataCollatorIntegrationTest(unittest.TestCase):
|
||||
|
||||
def test_nsp(self):
|
||||
tokenizer = BertTokenizer(self.vocab_file)
|
||||
features = [{"tokens_a": [0, 1, 2, 3, 4], "tokens_b": [0, 1, 2, 3, 4], "is_random_next": i} for i in range(2)]
|
||||
data_collator = DataCollatorForNextSentencePrediction(tokenizer)
|
||||
features = [
|
||||
{"input_ids": [0, 1, 2, 3, 4], "token_type_ids": [0, 1, 2, 3, 4], "next_sentence_label": i}
|
||||
for i in range(2)
|
||||
]
|
||||
data_collator = DataCollatorForLanguageModeling(tokenizer)
|
||||
batch = data_collator(features)
|
||||
|
||||
self.assertEqual(batch["input_ids"].shape, torch.Size((2, 512)))
|
||||
self.assertEqual(batch["token_type_ids"].shape, torch.Size((2, 512)))
|
||||
self.assertEqual(batch["labels"].shape, torch.Size((2, 512)))
|
||||
self.assertEqual(batch["input_ids"].shape, torch.Size((2, 5)))
|
||||
self.assertEqual(batch["token_type_ids"].shape, torch.Size((2, 5)))
|
||||
self.assertEqual(batch["labels"].shape, torch.Size((2, 5)))
|
||||
self.assertEqual(batch["next_sentence_label"].shape, torch.Size((2,)))
|
||||
|
||||
def test_sop(self):
|
||||
@@ -216,11 +217,11 @@ class DataCollatorIntegrationTest(unittest.TestCase):
|
||||
{
|
||||
"input_ids": torch.tensor([0, 1, 2, 3, 4]),
|
||||
"token_type_ids": torch.tensor([0, 1, 2, 3, 4]),
|
||||
"sentence_order_label": torch.tensor(i),
|
||||
"sentence_order_label": i,
|
||||
}
|
||||
for i in range(2)
|
||||
]
|
||||
data_collator = DataCollatorForSOP(tokenizer)
|
||||
data_collator = DataCollatorForLanguageModeling(tokenizer)
|
||||
batch = data_collator(features)
|
||||
|
||||
self.assertEqual(batch["input_ids"].shape, torch.Size((2, 5)))
|
||||
|
||||
@@ -24,8 +24,6 @@ from .test_modeling_common import ModelTesterMixin, ids_tensor, random_attention
|
||||
|
||||
|
||||
if is_torch_available():
|
||||
import torch
|
||||
|
||||
from transformers import (
|
||||
AlbertConfig,
|
||||
AlbertForMaskedLM,
|
||||
@@ -215,8 +213,6 @@ class AlbertModelTester:
|
||||
@require_torch
|
||||
class AlbertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
|
||||
test_gradient_checkpointing = True
|
||||
|
||||
all_model_classes = (
|
||||
(
|
||||
AlbertModel,
|
||||
@@ -267,28 +263,3 @@ class AlbertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
for model_name in ALBERT_PRETRAINED_MODEL_ARCHIVE_LIST[:1]:
|
||||
model = AlbertModel.from_pretrained(model_name)
|
||||
self.assertIsNotNone(model)
|
||||
|
||||
def test_model_gradient_checkpointing(self):
|
||||
if not self.test_gradient_checkpointing:
|
||||
return
|
||||
|
||||
for model_class in self.all_model_classes:
|
||||
print(model_class)
|
||||
config, inputs_dict = self.model_tester.prepare_config_and_inputs_for_common()
|
||||
model = model_class(config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
outputs_no_checkpointing = model(**self._prepare_for_class(inputs_dict, model_class))
|
||||
|
||||
config.gradient_checkpointing = True
|
||||
model_with_gc = model_class(config)
|
||||
model_with_gc.load_state_dict(model.state_dict())
|
||||
model_with_gc.to(torch_device)
|
||||
model_with_gc.eval()
|
||||
outputs_with_checkpointing = model_with_gc(**self._prepare_for_class(inputs_dict, model_class))
|
||||
|
||||
for output_no_checkpointing, output_with_checkpointing in zip(
|
||||
outputs_no_checkpointing, outputs_with_checkpointing
|
||||
):
|
||||
if isinstance(output_with_checkpointing, torch.Tensor):
|
||||
self.assertTrue(torch.allclose(output_no_checkpointing, output_with_checkpointing))
|
||||
@@ -140,7 +140,6 @@ class BARTModelTest(ModelTesterMixin, GenerationTesterMixin, unittest.TestCase):
|
||||
test_pruning = False
|
||||
test_head_masking = False
|
||||
test_missing_keys = False
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = ModelTester(self)
|
||||
|
||||
@@ -360,8 +360,6 @@ class BertModelTester:
|
||||
@require_torch
|
||||
class BertModelTest(ModelTesterMixin, GenerationTesterMixin, unittest.TestCase):
|
||||
|
||||
test_gradient_checkpointing = True
|
||||
|
||||
all_model_classes = (
|
||||
(
|
||||
BertModel,
|
||||
|
||||
@@ -189,9 +189,6 @@ class BertGenerationEncoderTest(ModelTesterMixin, GenerationTesterMixin, unittes
|
||||
all_model_classes = (BertGenerationEncoder, BertGenerationDecoder) if is_torch_available() else ()
|
||||
all_generative_model_classes = (BertGenerationDecoder,) if is_torch_available() else ()
|
||||
|
||||
# Should have a class in `AutoModelWithLMHead` to be tested on Gradient Checkpointing.
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = BertGenerationEncoderTester(self)
|
||||
self.config_tester = ConfigTester(self, config_class=BertGenerationConfig, hidden_size=37)
|
||||
|
||||
@@ -100,7 +100,6 @@ class BlenderbotTesterMixin(ModelTesterMixin, unittest.TestCase):
|
||||
test_pruning = False
|
||||
test_missing_keys = False
|
||||
test_torchscript = False
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = BlenderbotModelTester(self)
|
||||
|
||||
@@ -67,7 +67,6 @@ class ModelTesterMixin:
|
||||
test_head_masking = True
|
||||
test_missing_keys = True
|
||||
is_encoder_decoder = False
|
||||
test_gradient_checkpointing = True
|
||||
|
||||
def _prepare_for_class(self, inputs_dict, model_class, return_labels=False):
|
||||
inputs_dict = copy.deepcopy(inputs_dict)
|
||||
@@ -882,71 +881,6 @@ class ModelTesterMixin:
|
||||
with torch.no_grad():
|
||||
model(**inputs)[0]
|
||||
|
||||
def test_model_gradient_checkpointing_equivalent_results(self):
|
||||
if not self.test_gradient_checkpointing:
|
||||
return
|
||||
|
||||
for model_class in self.all_model_classes:
|
||||
print(model_class)
|
||||
config, inputs_dict = self.model_tester.prepare_config_and_inputs_for_common()
|
||||
model = model_class(config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
outputs_no_checkpointing = model(**self._prepare_for_class(inputs_dict, model_class))
|
||||
|
||||
config.gradient_checkpointing = True
|
||||
model_with_gc = model_class(config)
|
||||
model_with_gc.load_state_dict(model.state_dict())
|
||||
model_with_gc.to(torch_device)
|
||||
model_with_gc.eval()
|
||||
outputs_with_checkpointing = model_with_gc(**self._prepare_for_class(inputs_dict, model_class))
|
||||
|
||||
for output_no_checkpointing, output_with_checkpointing in zip(
|
||||
outputs_no_checkpointing, outputs_with_checkpointing
|
||||
):
|
||||
if isinstance(output_with_checkpointing, torch.Tensor):
|
||||
self.assertTrue(torch.allclose(output_no_checkpointing, output_with_checkpointing))
|
||||
|
||||
def test_model_gradient_checkpointing_memory(self):
|
||||
if not self.test_gradient_checkpointing:
|
||||
return
|
||||
|
||||
from transformers import PyTorchBenchmark, PyTorchBenchmarkArguments
|
||||
|
||||
config, inputs_dict = self.model_tester.prepare_config_and_inputs_for_common()
|
||||
gc_config = config
|
||||
gc_config.gradient_checkpointing = True
|
||||
|
||||
ngc_config, inputs_dict = self.model_tester.prepare_config_and_inputs_for_common()
|
||||
|
||||
assert gc_config.gradient_checkpointing and not ngc_config.gradient_checkpointing
|
||||
|
||||
# Gradient Checkpointing becomes really valuable for large input sizes.
|
||||
batch_size = int(self.model_tester.batch_size / 2)
|
||||
sequence_length = int(self.model_tester.max_position_embeddings / 2)
|
||||
|
||||
# Run the benchmark
|
||||
args = PyTorchBenchmarkArguments(
|
||||
models=["gc", "ngc"],
|
||||
batch_sizes=[batch_size],
|
||||
sequence_lengths=[sequence_length],
|
||||
inference=False,
|
||||
training=True,
|
||||
speed=False,
|
||||
)
|
||||
benchmark = PyTorchBenchmark(args, configs=[gc_config, ngc_config])
|
||||
|
||||
# Parse results
|
||||
result = benchmark.run()
|
||||
memory_result = result.memory_train_result
|
||||
gc_memory_result = memory_result["gc"]["result"][batch_size][sequence_length]
|
||||
ngc_memory_result = memory_result["ngc"]["result"][batch_size][sequence_length]
|
||||
|
||||
self.assertTrue(
|
||||
ngc_memory_result > gc_memory_result * 1.2,
|
||||
"Assert that gradient checkpointing requires a lot less memory.",
|
||||
)
|
||||
|
||||
@require_torch_multigpu
|
||||
def test_multigpu_data_parallel_forward(self):
|
||||
config, inputs_dict = self.model_tester.prepare_config_and_inputs_for_common()
|
||||
|
||||
@@ -54,9 +54,6 @@ class DebertaModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
test_head_masking = False
|
||||
is_encoder_decoder = False
|
||||
|
||||
# Needs to have DebertaForMaskedLM to test this.
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
class DebertaModelTester(object):
|
||||
def __init__(
|
||||
self,
|
||||
|
||||
@@ -211,7 +211,6 @@ class DistilBertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
test_torchscript = True
|
||||
test_resize_embeddings = True
|
||||
test_head_masking = True
|
||||
test_gradient_checkpointing = True
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = DistilBertModelTester(self)
|
||||
|
||||
@@ -329,8 +329,6 @@ class FlaubertModelTester(object):
|
||||
@require_torch
|
||||
class FlaubertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
all_model_classes = (
|
||||
(
|
||||
FlaubertModel,
|
||||
|
||||
@@ -128,7 +128,6 @@ class FSMTModelTest(ModelTesterMixin, GenerationTesterMixin, unittest.TestCase):
|
||||
test_pruning = False
|
||||
test_head_masking = False
|
||||
test_missing_keys = False
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = ModelTester(self)
|
||||
|
||||
@@ -454,28 +454,3 @@ class FunnelModelIntegrationTest(unittest.TestCase):
|
||||
expected_output_mean = torch.tensor(0.0256)
|
||||
self.assertTrue(torch.allclose(output.sum(), expected_output_sum, atol=1e-4))
|
||||
self.assertTrue(torch.allclose(output.mean(), expected_output_mean, atol=1e-4))
|
||||
|
||||
def test_model_gradient_checkpointing_equivalent_results(self):
|
||||
if not self.test_gradient_checkpointing:
|
||||
return
|
||||
|
||||
for model_class in self.all_model_classes:
|
||||
print(model_class)
|
||||
config, inputs_dict = self.model_tester.prepare_config_and_inputs_for_common()
|
||||
model = model_class(config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
outputs_no_checkpointing = model(**self._prepare_for_class(inputs_dict, model_class))
|
||||
|
||||
config.gradient_checkpointing = True
|
||||
model_with_gc = model_class(config)
|
||||
model_with_gc.load_state_dict(model.state_dict())
|
||||
model_with_gc.to(torch_device)
|
||||
model_with_gc.eval()
|
||||
outputs_with_checkpointing = model_with_gc(**self._prepare_for_class(inputs_dict, model_class))
|
||||
|
||||
for output_no_checkpointing, output_with_checkpointing in zip(
|
||||
outputs_no_checkpointing, outputs_with_checkpointing
|
||||
):
|
||||
if isinstance(output_with_checkpointing, torch.Tensor):
|
||||
self.assertTrue(torch.allclose(output_no_checkpointing, output_with_checkpointing))
|
||||
@@ -274,7 +274,6 @@ class LongformerModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
test_pruning = False # pruning is not supported
|
||||
test_headmasking = False # head masking is not supported
|
||||
test_torchscript = False
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
all_model_classes = (
|
||||
(
|
||||
|
||||
@@ -528,7 +528,10 @@ class LxmertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
test_head_masking = False
|
||||
test_pruning = False
|
||||
test_torchscript = False
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
test_head_masking = False
|
||||
test_pruning = False
|
||||
test_torchscript = False
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = LxmertModelTester(self)
|
||||
|
||||
@@ -862,7 +862,6 @@ class ProphetNetModelTest(ModelTesterMixin, GenerationTesterMixin, unittest.Test
|
||||
test_resize_embeddings = False
|
||||
test_headmasking = False
|
||||
is_encoder_decoder = True
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = ProphetNetModelTester(self)
|
||||
|
||||
@@ -583,7 +583,6 @@ class ReformerLocalAttnModelTest(ReformerTesterMixin, GenerationTesterMixin, Mod
|
||||
test_pruning = False
|
||||
test_headmasking = False
|
||||
test_torchscript = False
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
def prepare_kwargs(self):
|
||||
return {
|
||||
|
||||
@@ -232,7 +232,6 @@ class SqueezeBertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
test_torchscript = True
|
||||
test_resize_embeddings = True
|
||||
test_head_masking = False
|
||||
test_gradient_checkpointing = False
|
||||
|
||||
def setUp(self):
|
||||
self.model_tester = SqueezeBertModelTester(self)
|
||||
|
||||
Reference in new issue
Block a user