Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
acfcbc8833 | ||
|
|
f0bda06f43 | ||
|
|
c3c61ea017 | ||
|
|
45addfe96d | ||
|
|
7096e47513 | ||
|
|
ce374ba877 | ||
|
|
0a19a49dfe | ||
|
|
443b0cad96 | ||
|
|
74843695eb | ||
|
|
0befb51327 | ||
|
|
dc31a72f50 |
@@ -51,11 +51,4 @@ jobs:
|
||||
USE_CUDA: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 2 --dist=loadfile -s ./tests/ | tee output.txt
|
||||
- name: cat output.txt
|
||||
run: cat output.txt
|
||||
- name: Upload output.txt
|
||||
uses: actions/upload-artifact@v1
|
||||
with:
|
||||
name: pytest_output
|
||||
path: output.txt
|
||||
python -m pytest -n 2 --dist=loadfile -s ./tests/
|
||||
|
||||
@@ -46,11 +46,4 @@ jobs:
|
||||
USE_CUDA: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ | tee output.txt
|
||||
- name: cat output.txt
|
||||
run: cat output.txt
|
||||
- name: Upload output.txt
|
||||
uses: actions/upload-artifact@v1
|
||||
with:
|
||||
name: pytest_output
|
||||
path: output.txt
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/
|
||||
|
||||
@@ -61,6 +61,13 @@ FlaubertForSequenceClassification
|
||||
:members:
|
||||
|
||||
|
||||
FlaubertForTokenClassification
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.FlaubertForTokenClassification
|
||||
:members:
|
||||
|
||||
|
||||
FlaubertForQuestionAnsweringSimple
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
@@ -114,4 +121,4 @@ TFFlaubertForQuestionAnsweringSimple
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TFFlaubertForQuestionAnsweringSimple
|
||||
:members:
|
||||
:members:
|
||||
|
||||
@@ -108,11 +108,11 @@ any other model from the model hub):
|
||||
>>> model_name = "nlptown/bert-base-multilingual-uncased-sentiment"
|
||||
>>> model = AutoModelForSequenceClassification.from_pretrained(model_name)
|
||||
>>> tokenizer = AutoTokenizer.from_pretrained(model_name)
|
||||
>>> pipe = pipeline('sentiment-analysis', model=model, tokenizer=tokenizer)
|
||||
>>> classifier = pipeline('sentiment-analysis', model=model, tokenizer=tokenizer)
|
||||
>>> ## TENSORFLOW CODE
|
||||
>>> model_name = "nlptown/bert-base-multilingual-uncased-sentiment"
|
||||
>>> # This model only exists in PyTorch, so we use the `from_pt` flag to import that model in TensorFlow.
|
||||
>>> model = TFAutoModelForSequenceClassification.from_pretrained(model_name, from_pt=True)
|
||||
>>> model = TFAutoModelForSequenceClassification.from_pretrained(model_name, from_pt=True)
|
||||
>>> tokenizer = AutoTokenizer.from_pretrained(model_name)
|
||||
>>> classifier = pipeline('sentiment-analysis', model=model, tokenizer=tokenizer)
|
||||
|
||||
@@ -191,7 +191,7 @@ and get tensors back. You can specify all of that to the tokenizer:
|
||||
... return_tensors="tf"
|
||||
... )
|
||||
|
||||
The padding is automatically applied on the side the model expect it (in this case, on the right), with the
|
||||
The padding is automatically applied on the side expected by the model (in this case, on the right), with the
|
||||
padding token the model was pretrained with. The attention mask is also adapted to take the padding into account:
|
||||
|
||||
.. code-block::
|
||||
@@ -212,9 +212,9 @@ You can learn more about tokenizers :doc:`here <preprocessing>`.
|
||||
Using the model
|
||||
^^^^^^^^^^^^^^^
|
||||
|
||||
Once your input has been preprocessed by the tokenizer, you can directly send it to the model. As we mentioned, it will
|
||||
contain all the relevant information the model needs. If you're using a TensorFlow model, you can directly pass the
|
||||
dictionary keys to tensor, for a PyTorch model, you need to unpack the dictionary by adding :obj:`**`.
|
||||
Once your input has been preprocessed by the tokenizer, you can send it directly to the model. As we mentioned, it will
|
||||
contain all the relevant information the model needs. If you're using a TensorFlow model, you can pass the
|
||||
dictionary keys directly to tensor, for a PyTorch model, you need to unpack the dictionary by adding :obj:`**`.
|
||||
|
||||
.. code-block::
|
||||
|
||||
@@ -285,7 +285,7 @@ training loop. 🤗 Transformers also provides a :class:`~transformers.Trainer`
|
||||
you are using TensorFlow) class to help with your training (taking care of things such as distributed training, mixed
|
||||
precision, etc.). See the :doc:`training tutorial <training>` for more details.
|
||||
|
||||
Once your model is fine-tuned, you can save it with its tokenizer the following way:
|
||||
Once your model is fine-tuned, you can save it with its tokenizer in the following way:
|
||||
|
||||
::
|
||||
|
||||
@@ -329,7 +329,9 @@ pretrained model. Behind the scenes, the library has one model class per combina
|
||||
code is easy to access and tweak if you need to.
|
||||
|
||||
In our previous example, the model was called "distilbert-base-uncased-finetuned-sst-2-english", which means it's
|
||||
using the :doc:`DistilBERT </model_doc/distilbert>` architecture. The model automatically created is then a
|
||||
using the :doc:`DistilBERT </model_doc/distilbert>` architecture. As
|
||||
:class:`~transformers.AutoModelForSequenceClassification` (or :class:`~transformers.TFAutoModelForSequenceClassification`
|
||||
if you are using TensorFlow)` was used, the model automatically created is then a
|
||||
:class:`~transformers.DistilBertForSequenceClassification`. You can look at its documentation for all details relevant
|
||||
to that specific model, or browse the source code. This is how you would directly instantiate model and tokenizer
|
||||
without the auto magic:
|
||||
@@ -352,7 +354,7 @@ Customizing the model
|
||||
|
||||
If you want to change how the model itself is built, you can define your custom configuration class. Each architecture
|
||||
comes with its own relevant configuration (in the case of DistilBERT, :class:`~transformers.DistilBertConfig`) which
|
||||
allows you to specify any of the hidden dimension, dropout rate etc. If you do core modifications, like changing the
|
||||
allows you to specify any of the hidden dimension, dropout rate, etc. If you do core modifications, like changing the
|
||||
hidden size, you won't be able to use a pretrained model anymore and will need to train from scratch. You would then
|
||||
instantiate the model directly from this configuration.
|
||||
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
## About
|
||||
|
||||
The *french-postag-model* is a part of speech tagging model for French that was trained on the *free-french-treebank* dataset available on
|
||||
[github](https://github.com/nicolashernandez/free-french-treebank). The base tokenizer and model used for training is *'bert-base-multilingual-cased'*.
|
||||
|
||||
## Supported Tags
|
||||
|
||||
It uses the following tags:
|
||||
|
||||
| Tag | Category | Extra Info |
|
||||
|----------|:------------------------------:|------------:|
|
||||
| ADJ | adjectif | |
|
||||
| ADJWH | adjectif | |
|
||||
| ADV | adverbe | |
|
||||
| ADVWH | adverbe | |
|
||||
| CC | conjonction de coordination | |
|
||||
| CLO | pronom | obj |
|
||||
| CLR | pronom | refl |
|
||||
| CLS | pronom | suj |
|
||||
| CS | conjonction de subordination | |
|
||||
| DET | déterminant | |
|
||||
| DETWH | déterminant | |
|
||||
| ET | mot étranger | |
|
||||
| I | interjection | |
|
||||
| NC | nom commun | |
|
||||
| NPP | nom propre | |
|
||||
| P | préposition | |
|
||||
| P+D | préposition + déterminant | |
|
||||
| PONCT | signe de ponctuation | |
|
||||
| PREF | préfixe | |
|
||||
| PRO | autres pronoms | |
|
||||
| PROREL | autres pronoms | rel |
|
||||
| PROWH | autres pronoms | int |
|
||||
| U | ? | |
|
||||
| V | verbe | |
|
||||
| VIMP | verbe imperatif | |
|
||||
| VINF | verbe infinitif | |
|
||||
| VPP | participe passé | |
|
||||
| VPR | participe présent | |
|
||||
| VS | subjonctif | |
|
||||
|
||||
More information on the tags can be found here:
|
||||
|
||||
http://alpage.inria.fr/statgram/frdep/Publications/crabbecandi-taln2008-final.pdf
|
||||
|
||||
## Usage
|
||||
|
||||
The usage of this model follows the common transformers patterns. Here is a short example of its usage:
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModelForTokenClassification
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("gilf/french-postag-model")
|
||||
model = AutoModelForTokenClassification.from_pretrained("gilf/french-postag-model")
|
||||
|
||||
from transformers import pipeline
|
||||
|
||||
nlp_token_class = pipeline('ner', model=model, tokenizer=tokenizer, grouped_entities=True)
|
||||
|
||||
nlp_token_class('Face à un choc inédit, les mesures mises en place par le gouvernement ont permis une protection forte et efficace des ménages')
|
||||
```
|
||||
|
||||
The lines above would display something like this on a Jupyter notebook:
|
||||
|
||||
```
|
||||
[{'entity_group': 'PONCT', 'score': 0.0742340236902237, 'word': '[CLS]'},
|
||||
{'entity_group': 'U', 'score': 0.9995399713516235, 'word': 'Face'},
|
||||
{'entity_group': 'P', 'score': 0.9999609589576721, 'word': 'à'},
|
||||
{'entity_group': 'DET', 'score': 0.9999597072601318, 'word': 'un'},
|
||||
{'entity_group': 'NC', 'score': 0.9998948276042938, 'word': 'choc'},
|
||||
{'entity_group': 'ADJ', 'score': 0.995318204164505, 'word': 'inédit'},
|
||||
{'entity_group': 'PONCT', 'score': 0.9999793171882629, 'word': ','},
|
||||
{'entity_group': 'DET', 'score': 0.999964714050293, 'word': 'les'},
|
||||
{'entity_group': 'NC', 'score': 0.999936580657959, 'word': 'mesures'},
|
||||
{'entity_group': 'VPP', 'score': 0.9995776414871216, 'word': 'mises'},
|
||||
{'entity_group': 'P', 'score': 0.99996417760849, 'word': 'en'},
|
||||
{'entity_group': 'NC', 'score': 0.999882161617279, 'word': 'place'},
|
||||
{'entity_group': 'P', 'score': 0.9999671578407288, 'word': 'par'},
|
||||
{'entity_group': 'DET', 'score': 0.9999637603759766, 'word': 'le'},
|
||||
{'entity_group': 'NC', 'score': 0.9999350309371948, 'word': 'gouvernement'},
|
||||
{'entity_group': 'V', 'score': 0.9999298453330994, 'word': 'ont'},
|
||||
{'entity_group': 'VPP', 'score': 0.9998740553855896, 'word': 'permis'},
|
||||
{'entity_group': 'DET', 'score': 0.9999625086784363, 'word': 'une'},
|
||||
{'entity_group': 'NC', 'score': 0.9999420046806335, 'word': 'protection'},
|
||||
{'entity_group': 'ADJ', 'score': 0.9998913407325745, 'word': 'forte'},
|
||||
{'entity_group': 'CC', 'score': 0.9998615980148315, 'word': 'et'},
|
||||
{'entity_group': 'ADJ', 'score': 0.9998483657836914, 'word': 'efficace'},
|
||||
{'entity_group': 'P+D', 'score': 0.9987645149230957, 'word': 'des'},
|
||||
{'entity_group': 'NC', 'score': 0.8720395267009735, 'word': 'ménages [SEP]'}]
|
||||
```
|
||||
@@ -0,0 +1,46 @@
|
||||
## CodeBERT-base-mlm
|
||||
Pretrained weights for [CodeBERT: A Pre-Trained Model for Programming and Natural Languages](https://arxiv.org/abs/2002.08155).
|
||||
|
||||
### Training Data
|
||||
The model is trained on the code corpus of [CodeSearchNet](https://github.com/github/CodeSearchNet)
|
||||
|
||||
### Training Objective
|
||||
This model is initialized with Roberta-base and trained with a simple MLM (Masked Language Model) objective.
|
||||
|
||||
### Usage
|
||||
```python
|
||||
from transformers import RobertaTokenizer, RobertaForMaskedLM, pipeline
|
||||
|
||||
model = RobertaForMaskedLM.from_pretrained('microsoft/codebert-base-mlm')
|
||||
tokenizer = RobertaTokenizer.from_pretrained('microsoft/codebert-base-mlm')
|
||||
|
||||
code_example = "if (x is not None) <mask> (x>1)"
|
||||
fill_mask = pipeline('fill-mask', model=model, tokenizer=tokenizer)
|
||||
|
||||
outputs = fill_mask(code_example)
|
||||
print(outputs)
|
||||
```
|
||||
Expected results:
|
||||
```
|
||||
{'sequence': '<s> if (x is not None) and (x>1)</s>', 'score': 0.6049249172210693, 'token': 8}
|
||||
{'sequence': '<s> if (x is not None) or (x>1)</s>', 'score': 0.30680200457572937, 'token': 50}
|
||||
{'sequence': '<s> if (x is not None) if (x>1)</s>', 'score': 0.02133703976869583, 'token': 114}
|
||||
{'sequence': '<s> if (x is not None) then (x>1)</s>', 'score': 0.018607674166560173, 'token': 172}
|
||||
{'sequence': '<s> if (x is not None) AND (x>1)</s>', 'score': 0.007619690150022507, 'token': 4248}
|
||||
```
|
||||
|
||||
### Reference
|
||||
1. [Bimodal CodeBERT trained with MLM+RTD objective](https://huggingface.co/microsoft/codebert-base) (suitable for code search and document generation)
|
||||
2. 🤗 [Hugging Face's CodeBERTa](https://huggingface.co/huggingface/CodeBERTa-small-v1) (small size, 6 layers)
|
||||
|
||||
### Citation
|
||||
```bibtex
|
||||
@misc{feng2020codebert,
|
||||
title={CodeBERT: A Pre-Trained Model for Programming and Natural Languages},
|
||||
author={Zhangyin Feng and Daya Guo and Duyu Tang and Nan Duan and Xiaocheng Feng and Ming Gong and Linjun Shou and Bing Qin and Ting Liu and Daxin Jiang and Ming Zhou},
|
||||
year={2020},
|
||||
eprint={2002.08155},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,27 @@
|
||||
## CodeBERT-base
|
||||
Pretrained weights for [CodeBERT: A Pre-Trained Model for Programming and Natural Languages](https://arxiv.org/abs/2002.08155).
|
||||
|
||||
### Training Data
|
||||
The model is trained on bi-modal data (documents & code) of [CodeSearchNet](https://github.com/github/CodeSearchNet)
|
||||
|
||||
### Training Objective
|
||||
This model is initialized with Roberta-base and trained with MLM+RTD objective (cf. the paper).
|
||||
|
||||
### Usage
|
||||
Please see [the official repository](https://github.com/microsoft/CodeBERT) for scripts that support "code search" and "code-to-document generation".
|
||||
|
||||
### Reference
|
||||
1. [CodeBERT trained with Masked LM objective](https://huggingface.co/microsoft/codebert-base-mlm) (suitable for code completion)
|
||||
2. 🤗 [Hugging Face's CodeBERTa](https://huggingface.co/huggingface/CodeBERTa-small-v1) (small size, 6 layers)
|
||||
|
||||
### Citation
|
||||
```bibtex
|
||||
@misc{feng2020codebert,
|
||||
title={CodeBERT: A Pre-Trained Model for Programming and Natural Languages},
|
||||
author={Zhangyin Feng and Daya Guo and Duyu Tang and Nan Duan and Xiaocheng Feng and Ming Gong and Linjun Shou and Bing Qin and Ting Liu and Daxin Jiang and Ming Zhou},
|
||||
year={2020},
|
||||
eprint={2002.08155},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL}
|
||||
}
|
||||
```
|
||||
@@ -65,6 +65,7 @@ from .data import (
|
||||
xnli_processors,
|
||||
xnli_tasks_num_labels,
|
||||
)
|
||||
|
||||
# Files and general utilities
|
||||
from .file_utils import (
|
||||
CONFIG_NAME,
|
||||
@@ -86,8 +87,10 @@ from .file_utils import (
|
||||
is_torch_tpu_available,
|
||||
)
|
||||
from .hf_argparser import HfArgumentParser
|
||||
|
||||
# Model Cards
|
||||
from .modelcard import ModelCard
|
||||
|
||||
# TF 2.0 <=> PyTorch conversion utilities
|
||||
from .modeling_tf_pytorch_utils import (
|
||||
convert_tf_weight_name_to_pt_weight_name,
|
||||
@@ -98,6 +101,7 @@ from .modeling_tf_pytorch_utils import (
|
||||
load_tf2_model_in_pytorch_model,
|
||||
load_tf2_weights_in_pytorch_model,
|
||||
)
|
||||
|
||||
# Pipelines
|
||||
from .pipelines import (
|
||||
CsvPipelineDataFormat,
|
||||
@@ -116,6 +120,7 @@ from .pipelines import (
|
||||
TranslationPipeline,
|
||||
pipeline,
|
||||
)
|
||||
|
||||
# Tokenizers
|
||||
from .tokenization_albert import AlbertTokenizer
|
||||
from .tokenization_auto import TOKENIZER_MAPPING, AutoTokenizer
|
||||
@@ -157,6 +162,7 @@ from .tokenization_utils_fast import PreTrainedTokenizerFast
|
||||
from .tokenization_xlm import XLMTokenizer
|
||||
from .tokenization_xlm_roberta import XLMRobertaTokenizer
|
||||
from .tokenization_xlnet import SPIECE_UNDERLINE, XLNetTokenizer
|
||||
|
||||
# Trainer
|
||||
from .trainer_utils import EvalPrediction, set_seed
|
||||
from .training_args import TrainingArguments
|
||||
@@ -347,6 +353,7 @@ if is_torch_available():
|
||||
FlaubertModel,
|
||||
FlaubertWithLMHeadModel,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
FlaubertForQuestionAnswering,
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
|
||||
@@ -14,6 +14,7 @@ import sys
|
||||
import tarfile
|
||||
import tempfile
|
||||
from contextlib import contextmanager
|
||||
from dataclasses import fields
|
||||
from functools import partial, wraps
|
||||
from hashlib import sha256
|
||||
from pathlib import Path
|
||||
@@ -856,12 +857,83 @@ def tf_required(func):
|
||||
return wrapper
|
||||
|
||||
|
||||
class ModelOutput:
|
||||
"""
|
||||
Base class for all model outputs as dataclass. Has a ``__getitem__`` that allows indexing by integer or slice (like
|
||||
a tuple) or strings (like a dictionnary) that will ignore the ``None`` attributes.
|
||||
class ModelOutput(dict):
|
||||
""" Base class for all model outputs as dataclass.
|
||||
|
||||
The sub-classes of ``ModelOutput`` must have **at most one** required field (the first field)
|
||||
(see ``__post_init__``docstring for details)
|
||||
|
||||
``ModelOutput`` has a ``__getitem__`` method that allows indexing by:
|
||||
- integer and slice (like a tuple) or
|
||||
- strings (like a dictionnary).
|
||||
|
||||
``__getitem__`` will ignore attributes containing ``None`` values when indexing with integers.
|
||||
|
||||
Apart from accepting integers as key, ``ModelOutput``mostly behaves like a dictionnary for compatiblity with torch.DataParallel:
|
||||
- Sub-class of ``dict``
|
||||
- providing an iterator of (keys, values) to the first argument without any other argument will set the associated attributes
|
||||
- ``__setitem__``, ``__delitem__``, ``setdefault``, ``pop``, `ùpdate`` will raise errors
|
||||
- ``__iter__`` iterate over the keys
|
||||
|
||||
"""
|
||||
|
||||
def __post_init__(self):
|
||||
""" Does a few safety checks.
|
||||
|
||||
If the first field is a list/tuple, spread its values in the other fields.
|
||||
This is currently necessary for compatibility with torch.DataParallel which only handles dict/list/tuples
|
||||
cf. https://github.com/huggingface/transformers/issues/5693
|
||||
and https://github.com/pytorch/pytorch/issues/41327
|
||||
"""
|
||||
class_fields = fields(self)
|
||||
|
||||
# Safety and consistency checks
|
||||
assert len(class_fields), f"{self.__class__.__name__} has no fields."
|
||||
assert all(
|
||||
field.default is None for field in class_fields[1:]
|
||||
), f"{self.__class__.__name__} should not have more than one required field."
|
||||
|
||||
# Check if we should spread the first field on the other fields in a dict-mapping fashion
|
||||
first_field = getattr(self, class_fields[0].name)
|
||||
other_fields_are_none = all(getattr(self, field.name) is None for field in class_fields[1:])
|
||||
|
||||
if other_fields_are_none:
|
||||
try:
|
||||
iterator = iter(first_field)
|
||||
first_field_iterator = True
|
||||
except TypeError:
|
||||
first_field_iterator = False
|
||||
|
||||
# if we provided an iterator as first field and the iterator is a (key, value) iterator
|
||||
# set the associated fields
|
||||
if first_field_iterator:
|
||||
for element in iterator:
|
||||
if (
|
||||
not isinstance(element, (list, tuple))
|
||||
or not len(element) == 2
|
||||
or not isinstance(element[0], str)
|
||||
):
|
||||
break
|
||||
setattr(self, element[0], element[1])
|
||||
|
||||
def __delitem__(self, *args, **kwargs):
|
||||
raise Exception(f"You cannot use ``__delitem__`` on a {self.__class__.__name__} instance.")
|
||||
|
||||
def setdefault(self, *args, **kwargs):
|
||||
raise Exception(f"You cannot use ``setdefault`` on a {self.__class__.__name__} instance.")
|
||||
|
||||
def pop(self, *args, **kwargs):
|
||||
raise Exception(f"You cannot use ``pop`` on a {self.__class__.__name__} instance.")
|
||||
|
||||
def update(self, *args, **kwargs):
|
||||
raise Exception(f"You cannot use ``update`` on a {self.__class__.__name__} instance.")
|
||||
|
||||
def __iter__(self):
|
||||
""" Will return the attributes names (dictionnary-like) """
|
||||
for f in self.__dataclass_fields__.keys():
|
||||
if getattr(self, f, None) is not None:
|
||||
yield f
|
||||
|
||||
def to_tuple(self):
|
||||
"""
|
||||
Converts :obj:`self` to a tuple.
|
||||
@@ -881,5 +953,11 @@ class ModelOutput:
|
||||
def __getitem__(self, i):
|
||||
return self.to_dict()[i] if isinstance(i, str) else self.to_tuple()[i]
|
||||
|
||||
def __setitem__(self, key, value):
|
||||
if isinstance(key, str):
|
||||
setattr(self, key, value)
|
||||
else:
|
||||
raise Exception(f"Key {key} must be a string but was given a {type(key)}.")
|
||||
|
||||
def __len__(self):
|
||||
return len(self.to_tuple())
|
||||
|
||||
@@ -430,9 +430,9 @@ class AlbertForPretrainingOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
prediction_logits: torch.FloatTensor
|
||||
sop_logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
prediction_logits: torch.FloatTensor = None
|
||||
sop_logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
|
||||
@@ -100,6 +100,7 @@ from .modeling_encoder_decoder import EncoderDecoderModel
|
||||
from .modeling_flaubert import (
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
FlaubertModel,
|
||||
FlaubertWithLMHeadModel,
|
||||
)
|
||||
@@ -326,6 +327,7 @@ MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING = OrderedDict(
|
||||
[
|
||||
(DistilBertConfig, DistilBertForTokenClassification),
|
||||
(CamembertConfig, CamembertForTokenClassification),
|
||||
(FlaubertConfig, FlaubertForTokenClassification),
|
||||
(XLMConfig, XLMForTokenClassification),
|
||||
(XLMRobertaConfig, XLMRobertaForTokenClassification),
|
||||
(LongformerConfig, LongformerForTokenClassification),
|
||||
@@ -1552,6 +1554,7 @@ class AutoModelForTokenClassification:
|
||||
- isInstance of `bert` configuration class: :class:`~transformers.BertModelForTokenClassification` (Bert model)
|
||||
- isInstance of `albert` configuration class: :class:`~transformers.AlbertForTokenClassification` (AlBert model)
|
||||
- isInstance of `xlnet` configuration class: :class:`~transformers.XLNetModelForTokenClassification` (XLNet model)
|
||||
- isInstance of `flaubert` configuration class: :class:`~transformers.FlaubertForTokenClassification` (Flaubert model)
|
||||
- isInstance of `camembert` configuration class: :class:`~transformers.CamembertModelForTokenClassification` (Camembert model)
|
||||
- isInstance of `roberta` configuration class: :class:`~transformers.RobertaModelForTokenClassification` (Roberta model)
|
||||
- isInstance of `electra` configuration class: :class:`~transformers.ElectraForTokenClassification` (Electra model)
|
||||
@@ -1589,6 +1592,7 @@ class AutoModelForTokenClassification:
|
||||
- `camembert`: :class:`~transformers.CamembertForTokenClassification` (Camembert model)
|
||||
- `bert`: :class:`~transformers.BertForTokenClassification` (Bert model)
|
||||
- `xlnet`: :class:`~transformers.XLNetForTokenClassification` (XLNet model)
|
||||
- `flaubert`: :class:`~transformers.FlaubertForTokenClassification` (Flaubert model)
|
||||
- `roberta`: :class:`~transformers.RobertaForTokenClassification` (Roberta model)
|
||||
- `electra`: :class:`~transformers.ElectraForTokenClassification` (Electra model)
|
||||
|
||||
|
||||
@@ -605,9 +605,9 @@ class BertForPretrainingOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
prediction_logits: torch.FloatTensor
|
||||
seq_relationship_logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
prediction_logits: torch.FloatTensor = None
|
||||
seq_relationship_logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
|
||||
@@ -73,7 +73,7 @@ class DPRContextEncoderOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
pooler_output: torch.FloatTensor
|
||||
pooler_output: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -102,7 +102,7 @@ class DPRQuestionEncoderOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
pooler_output: torch.FloatTensor
|
||||
pooler_output: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -133,9 +133,9 @@ class DPRReaderOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
start_logits: torch.FloatTensor
|
||||
end_logits: torch.FloatTensor
|
||||
relevance_logits: torch.FloatTensor
|
||||
start_logits: torch.FloatTensor = None
|
||||
end_logits: torch.FloatTensor = None
|
||||
relevance_logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
|
||||
@@ -208,8 +208,8 @@ class ElectraForPretrainingOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
|
||||
@@ -28,6 +28,7 @@ from .modeling_xlm import (
|
||||
XLMForQuestionAnswering,
|
||||
XLMForQuestionAnsweringSimple,
|
||||
XLMForSequenceClassification,
|
||||
XLMForTokenClassification,
|
||||
XLMModel,
|
||||
XLMWithLMHeadModel,
|
||||
get_masks,
|
||||
@@ -326,6 +327,25 @@ class FlaubertForSequenceClassification(XLMForSequenceClassification):
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
"""Flaubert Model with a token classification head on top (a linear layer on top of
|
||||
the hidden-states output) e.g. for Named-Entity-Recognition (NER) tasks. """,
|
||||
FLAUBERT_START_DOCSTRING,
|
||||
)
|
||||
class FlaubertForTokenClassification(XLMForTokenClassification):
|
||||
"""
|
||||
This class overrides :class:`~transformers.XLMForTokenClassification`. Please check the
|
||||
superclass for the appropriate documentation alongside usage examples.
|
||||
"""
|
||||
|
||||
config_class = FlaubertConfig
|
||||
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
"""Flaubert Model with a span classification head on top for extractive question-answering tasks like SQuAD (a linear layers on top of
|
||||
the hidden-states output to compute `span start logits` and `span end logits`). """,
|
||||
|
||||
@@ -323,10 +323,10 @@ class GPT2DoubleHeadsModelOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
lm_loss: Optional[torch.FloatTensor]
|
||||
mc_loss: Optional[torch.FloatTensor]
|
||||
lm_logits: torch.FloatTensor
|
||||
mc_logits: torch.FloatTensor
|
||||
lm_loss: torch.FloatTensor = None
|
||||
mc_loss: torch.FloatTensor = None
|
||||
lm_logits: torch.FloatTensor = None
|
||||
mc_logits: torch.FloatTensor = None
|
||||
past_key_values: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -442,12 +442,14 @@ class LongformerSelfAttention(nn.Module):
|
||||
if output_attentions:
|
||||
if is_global_attn:
|
||||
# With global attention, return global attention probabilities only
|
||||
# batch_size x num_heads x max_num_global_attention_tokens x sequence_length
|
||||
# which is the attention weights from tokens with global attention to all tokens
|
||||
# It doesn't not return local attention
|
||||
# In case of variable number of global attantion in the rows of a batch,
|
||||
# attn_probs are padded with -10000.0 attention scores
|
||||
attn_probs = attn_probs.view(batch_size, self.num_heads, max_num_global_attn_indices, seq_len)
|
||||
# batch_size x num_heads x sequence_length x window_size
|
||||
# which is the attention weights from all tokens to all tokens for global attention
|
||||
# It doesn't not return local attention. Only tokens with global attention have values > 0.0
|
||||
attn_probs = attn_probs[:, :, :, :max_num_global_attn_indices]
|
||||
# pad attn_probs to max length with 0.0 since global attn did not attend there
|
||||
window_size = self.one_sided_attn_window_size * 2 + 1
|
||||
attn_probs = F.pad(attn_probs, (0, window_size - max_num_global_attn_indices), value=0.0,)
|
||||
attn_probs = attn_probs.permute(0, 2, 1, 3)
|
||||
else:
|
||||
# without global attention, return local attention probabilities
|
||||
# batch_size x num_heads x sequence_length x window_size
|
||||
|
||||
@@ -705,9 +705,9 @@ class MobileBertForPretrainingOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
prediction_logits: torch.FloatTensor
|
||||
seq_relationship_logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
prediction_logits: torch.FloatTensor = None
|
||||
seq_relationship_logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
|
||||
@@ -314,10 +314,10 @@ class OpenAIGPTDoubleHeadsModelOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
lm_loss: Optional[torch.FloatTensor]
|
||||
mc_loss: Optional[torch.FloatTensor]
|
||||
lm_logits: torch.FloatTensor
|
||||
mc_logits: torch.FloatTensor
|
||||
lm_loss: torch.FloatTensor = None
|
||||
mc_loss: torch.FloatTensor = None
|
||||
lm_logits: torch.FloatTensor = None
|
||||
mc_logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
|
||||
@@ -63,7 +63,7 @@ class BaseModelOutputWithPooling(ModelOutput):
|
||||
"""
|
||||
|
||||
last_hidden_state: torch.FloatTensor
|
||||
pooler_output: torch.FloatTensor
|
||||
pooler_output: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -178,8 +178,8 @@ class CausalLMOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -213,8 +213,8 @@ class CausalLMOutputWithPast(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
past_key_values: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
@@ -243,8 +243,8 @@ class MaskedLMOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -291,8 +291,8 @@ class Seq2SeqLMOutput(ModelOutput):
|
||||
self-attention heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
decoder_past_key_values: Optional[List[torch.FloatTensor]] = None
|
||||
decoder_hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
decoder_attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
@@ -324,8 +324,8 @@ class NextSentencePredictorOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -353,8 +353,8 @@ class SequenceClassifierOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -401,8 +401,8 @@ class Seq2SeqSequenceClassifierOutput(ModelOutput):
|
||||
self-attention heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
decoder_past_key_values: Optional[List[torch.FloatTensor]] = None
|
||||
decoder_hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
decoder_attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
@@ -436,8 +436,8 @@ class MultipleChoiceModelOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -465,8 +465,8 @@ class TokenClassifierOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -496,9 +496,9 @@ class QuestionAnsweringModelOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
start_logits: torch.FloatTensor
|
||||
end_logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
start_logits: torch.FloatTensor = None
|
||||
end_logits: torch.FloatTensor = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -547,9 +547,9 @@ class Seq2SeqQuestionAnsweringModelOutput(ModelOutput):
|
||||
self-attention heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
start_logits: torch.FloatTensor
|
||||
end_logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
start_logits: torch.FloatTensor = None
|
||||
end_logits: torch.FloatTensor = None
|
||||
decoder_past_key_values: Optional[List[torch.FloatTensor]] = None
|
||||
decoder_hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
decoder_attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -618,7 +618,7 @@ class TransfoXLModelOutput(ModelOutput):
|
||||
"""
|
||||
|
||||
last_hidden_state: torch.FloatTensor
|
||||
mems: List[torch.FloatTensor]
|
||||
mems: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -653,8 +653,8 @@ class TransfoXLLMHeadModelOutput(ModelOutput):
|
||||
"""
|
||||
|
||||
losses: Optional[torch.FloatTensor]
|
||||
prediction_scores: torch.FloatTensor
|
||||
mems: List[torch.FloatTensor]
|
||||
prediction_scores: torch.FloatTensor = None
|
||||
mems: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
|
||||
@@ -627,8 +627,8 @@ class XLNetLMHeadModelOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
mems: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
@@ -661,8 +661,8 @@ class XLNetForSequenceClassificationOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
mems: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
@@ -695,8 +695,8 @@ class XLNetForTokenClassificationOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
mems: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
@@ -731,8 +731,8 @@ class XLNetForMultipleChoiceOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
logits: torch.FloatTensor = None
|
||||
mems: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
@@ -767,9 +767,9 @@ class XLNetForQuestionAnsweringSimpleOutput(ModelOutput):
|
||||
heads.
|
||||
"""
|
||||
|
||||
loss: Optional[torch.FloatTensor]
|
||||
start_logits: torch.FloatTensor
|
||||
end_logits: torch.FloatTensor
|
||||
loss: torch.FloatTensor = None
|
||||
start_logits: torch.FloatTensor = None
|
||||
end_logits: torch.FloatTensor = None
|
||||
mems: Optional[List[torch.FloatTensor]] = None
|
||||
hidden_states: Optional[Tuple[torch.FloatTensor]] = None
|
||||
attentions: Optional[Tuple[torch.FloatTensor]] = None
|
||||
|
||||
@@ -46,6 +46,10 @@ if is_tf_available():
|
||||
TFAutoModelForQuestionAnswering,
|
||||
TFAutoModelForTokenClassification,
|
||||
TFAutoModelWithLMHead,
|
||||
TF_MODEL_WITH_LM_HEAD_MAPPING,
|
||||
TF_MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING,
|
||||
TF_MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING,
|
||||
TF_MODEL_FOR_QUESTION_ANSWERING_MAPPING,
|
||||
)
|
||||
|
||||
if is_torch_available():
|
||||
@@ -57,6 +61,11 @@ if is_torch_available():
|
||||
AutoModelForTokenClassification,
|
||||
AutoModelWithLMHead,
|
||||
AutoModelForSeq2SeqLM,
|
||||
MODEL_WITH_LM_HEAD_MAPPING,
|
||||
MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING,
|
||||
MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING,
|
||||
MODEL_FOR_QUESTION_ANSWERING_MAPPING,
|
||||
MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -396,6 +405,7 @@ class Pipeline(_ScikitCompat):
|
||||
if framework is None:
|
||||
framework = get_framework()
|
||||
|
||||
self.task = task
|
||||
self.model = model
|
||||
self.tokenizer = tokenizer
|
||||
self.modelcard = modelcard
|
||||
@@ -469,6 +479,19 @@ class Pipeline(_ScikitCompat):
|
||||
"""
|
||||
return {name: tensor.to(self.device) for name, tensor in inputs.items()}
|
||||
|
||||
def check_model_type(self, supported_models):
|
||||
"""
|
||||
Check if the model class is in the supported class list of the pipeline.
|
||||
"""
|
||||
if not isinstance(supported_models, list): # Create from a model mapping
|
||||
supported_models = [item[1].__name__ for item in supported_models.items()]
|
||||
if self.model.__class__.__name__ not in supported_models:
|
||||
raise PipelineException(
|
||||
self.task,
|
||||
self.model.base_model_prefix,
|
||||
f"The model '{self.model.__class__.__name__}' is not supported for {self.task}. Supported models are {supported_models}",
|
||||
)
|
||||
|
||||
def _parse_and_tokenize(self, *args, padding=True, add_special_tokens=True, **kwargs):
|
||||
"""
|
||||
Parse arguments and tokenize
|
||||
@@ -615,6 +638,11 @@ class TextGenerationPipeline(Pipeline):
|
||||
"TFCTRLLMHeadModel",
|
||||
]
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
self.check_model_type(self.ALLOWED_MODELS)
|
||||
|
||||
# overriding _parse_and_tokenize to allow for unusual language-modeling tokenizer arguments
|
||||
|
||||
def _parse_and_tokenize(self, *args, padding=True, add_special_tokens=True, **kwargs):
|
||||
@@ -640,12 +668,6 @@ class TextGenerationPipeline(Pipeline):
|
||||
def __call__(
|
||||
self, *args, return_tensors=False, return_text=True, clean_up_tokenization_spaces=False, **generate_kwargs
|
||||
):
|
||||
if self.model.__class__.__name__ not in self.ALLOWED_MODELS:
|
||||
raise NotImplementedError(
|
||||
"Generation is currently not supported for {}. Please select a model from {} for generation.".format(
|
||||
self.model.__class__.__name__, self.ALLOWED_MODELS
|
||||
)
|
||||
)
|
||||
|
||||
text_inputs = self._args_parser(*args)
|
||||
|
||||
@@ -771,6 +793,12 @@ class TextClassificationPipeline(Pipeline):
|
||||
def __init__(self, return_all_scores: bool = False, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING
|
||||
if self.framework == "tf"
|
||||
else MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING
|
||||
)
|
||||
|
||||
self.return_all_scores = return_all_scores
|
||||
|
||||
def __call__(self, *args, **kwargs):
|
||||
@@ -847,6 +875,8 @@ class FillMaskPipeline(Pipeline):
|
||||
task=task,
|
||||
)
|
||||
|
||||
self.check_model_type(TF_MODEL_WITH_LM_HEAD_MAPPING if self.framework == "tf" else MODEL_WITH_LM_HEAD_MAPPING)
|
||||
|
||||
self.topk = topk
|
||||
|
||||
def ensure_exactly_one_mask_token(self, masked_index: np.ndarray):
|
||||
@@ -980,6 +1010,12 @@ class TokenClassificationPipeline(Pipeline):
|
||||
task=task,
|
||||
)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING
|
||||
if self.framework == "tf"
|
||||
else MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING
|
||||
)
|
||||
|
||||
self._basic_tokenizer = BasicTokenizer(do_lower_case=False)
|
||||
self.ignore_labels = ignore_labels
|
||||
self.grouped_entities = grouped_entities
|
||||
@@ -1220,6 +1256,10 @@ class QuestionAnsweringPipeline(Pipeline):
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_FOR_QUESTION_ANSWERING_MAPPING if self.framework == "tf" else MODEL_FOR_QUESTION_ANSWERING_MAPPING
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def create_sample(
|
||||
question: Union[str, List[str]], context: Union[str, List[str]]
|
||||
@@ -1483,9 +1523,13 @@ class SummarizationPipeline(Pipeline):
|
||||
on the associated CUDA device id.
|
||||
"""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
def __init__(self, *args, **kwargs):
|
||||
kwargs.update(task="summarization")
|
||||
super().__init__(**kwargs)
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_WITH_LM_HEAD_MAPPING if self.framework == "tf" else MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING
|
||||
)
|
||||
|
||||
def __call__(
|
||||
self, *documents, return_tensors=False, return_text=True, clean_up_tokenization_spaces=False, **generate_kwargs
|
||||
@@ -1615,6 +1659,11 @@ class TranslationPipeline(Pipeline):
|
||||
on the associated CUDA device id.
|
||||
"""
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
self.check_model_type(TF_MODEL_WITH_LM_HEAD_MAPPING if self.framework == "tf" else MODEL_WITH_LM_HEAD_MAPPING)
|
||||
|
||||
def __call__(
|
||||
self, *args, return_tensors=False, return_text=True, clean_up_tokenization_spaces=False, **generate_kwargs
|
||||
):
|
||||
|
||||
@@ -61,7 +61,7 @@ PRETRAINED_POSITIONAL_EMBEDDINGS_SIZES = {
|
||||
|
||||
class T5Tokenizer(PreTrainedTokenizer):
|
||||
"""
|
||||
Constructs an XLNet tokenizer. Based on `SentencePiece <https://github.com/google/sentencepiece>`__ .
|
||||
Constructs a T5 tokenizer. Based on `SentencePiece <https://github.com/google/sentencepiece>`__ .
|
||||
|
||||
This tokenizer inherits from :class:`~transformers.PreTrainedTokenizer` which contains most of the methods. Users
|
||||
should refer to the superclass for more information regarding methods.
|
||||
|
||||
@@ -618,6 +618,9 @@ class Trainer:
|
||||
|
||||
if self.args.past_index >= 0 and self._past is not None:
|
||||
inputs["mems"] = self._past
|
||||
# Our model outputs do not work with DataParallel, so forcing return tuple.
|
||||
if self.args.n_gpu > 1:
|
||||
inputs["return_tuple"] = True
|
||||
|
||||
outputs = model(**inputs)
|
||||
loss = outputs[0] # model outputs are always tuple in transformers (see doc)
|
||||
@@ -818,6 +821,9 @@ class Trainer:
|
||||
inputs[k] = v.to(self.args.device)
|
||||
if self.args.past_index >= 0:
|
||||
inputs["mems"] = past
|
||||
# Our model outputs do not work with DataParallel, so forcing return tuple.
|
||||
if self.args.n_gpu > 1:
|
||||
inputs["return_tuple"] = True
|
||||
|
||||
with torch.no_grad():
|
||||
outputs = model(**inputs)
|
||||
|
||||
@@ -44,6 +44,7 @@ from utils_squad import (
|
||||
write_predictions,
|
||||
write_predictions_extended,
|
||||
)
|
||||
|
||||
# The follwing import is the official SQuAD evaluation script (2.0).
|
||||
# You can remove it from the dependencies if you are using this script outside of the library
|
||||
# We've added it here for automated tests (see examples/test_examples.py file)
|
||||
|
||||
@@ -21,6 +21,7 @@ import logging
|
||||
import math
|
||||
|
||||
from transformers.tokenization_bert import BasicTokenizer, whitespace_tokenize
|
||||
|
||||
# Required by XLNet evaluation method to compute optimal threshold (see write_predictions_extended() method)
|
||||
from utils_squad_evaluate import find_all_best_thresh_v2, get_raw_scores, make_qid_to_has_ans
|
||||
|
||||
|
||||
@@ -31,6 +31,7 @@ if is_torch_available():
|
||||
FlaubertForQuestionAnswering,
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
)
|
||||
from transformers.modeling_flaubert import FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST
|
||||
|
||||
@@ -294,6 +295,30 @@ class FlaubertModelTester(object):
|
||||
self.parent.assertListEqual(list(result["loss"].size()), [])
|
||||
self.parent.assertListEqual(list(result["logits"].size()), [self.batch_size, self.type_sequence_label_size])
|
||||
|
||||
def create_and_check_flaubert_token_classif(
|
||||
self,
|
||||
config,
|
||||
input_ids,
|
||||
token_type_ids,
|
||||
input_lengths,
|
||||
sequence_labels,
|
||||
token_labels,
|
||||
is_impossible_labels,
|
||||
input_mask,
|
||||
):
|
||||
config.num_labels = self.num_labels
|
||||
model = FlaubertForTokenClassification(config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
|
||||
loss, logits = model(input_ids, attention_mask=input_mask, labels=token_labels)
|
||||
result = {
|
||||
"loss": loss,
|
||||
"logits": logits,
|
||||
}
|
||||
self.parent.assertListEqual(list(result["logits"].size()), [self.batch_size, self.seq_length, self.num_labels])
|
||||
self.check_loss_output(result)
|
||||
|
||||
def prepare_config_and_inputs_for_common(self):
|
||||
config_and_inputs = self.prepare_config_and_inputs()
|
||||
(
|
||||
@@ -320,6 +345,7 @@ class FlaubertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
FlaubertForQuestionAnswering,
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
)
|
||||
if is_torch_available()
|
||||
else ()
|
||||
@@ -352,6 +378,10 @@ class FlaubertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_flaubert_sequence_classif(*config_and_inputs)
|
||||
|
||||
def test_flaubert_token_classif(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_flaubert_token_classif(*config_and_inputs)
|
||||
|
||||
@slow
|
||||
def test_model_from_pretrained(self):
|
||||
for model_name in FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST[:1]:
|
||||
|
||||
@@ -297,7 +297,7 @@ class XLMModelTester:
|
||||
self.parent.assertListEqual(list(result["loss"].size()), [])
|
||||
self.parent.assertListEqual(list(result["logits"].size()), [self.batch_size, self.type_sequence_label_size])
|
||||
|
||||
def create_and_check_xlm_for_token_classification(
|
||||
def create_and_check_xlm_token_classif(
|
||||
self,
|
||||
config,
|
||||
input_ids,
|
||||
@@ -383,9 +383,9 @@ class XLMModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_xlm_sequence_classif(*config_and_inputs)
|
||||
|
||||
def test_xlm_for_token_classification(self):
|
||||
def test_xlm_token_classif(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_xlm_for_token_classification(*config_and_inputs)
|
||||
self.model_tester.create_and_check_xlm_token_classif(*config_and_inputs)
|
||||
|
||||
@slow
|
||||
def test_model_from_pretrained(self):
|
||||
|
||||
Reference in New Issue
Block a user