Compare commits
37
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f634ba14df | ||
|
|
07d7ff6524 | ||
|
|
11269ec971 | ||
|
|
359d9516b0 | ||
|
|
79d78050e1 | ||
|
|
4507c41108 | ||
|
|
e412924db0 | ||
|
|
757a0a95a0 | ||
|
|
2f9976c101 | ||
|
|
d39b3b4cd4 | ||
|
|
d2cb489b0d | ||
|
|
0641e9ce5a | ||
|
|
ed52fea1fc | ||
|
|
dd9ad95644 | ||
|
|
8ab565a4be | ||
|
|
92dc959224 | ||
|
|
baf93b02c4 | ||
|
|
5d178954c9 | ||
|
|
ac921f0385 | ||
|
|
21c1fe5290 | ||
|
|
2db1cc807b | ||
|
|
dae244ad89 | ||
|
|
b2505f7db7 | ||
|
|
838950ee44 | ||
|
|
4d5a8d6557 | ||
|
|
cd30f98fd2 | ||
|
|
f867000f56 | ||
|
|
f0bda06f43 | ||
|
|
c3c61ea017 | ||
|
|
45addfe96d | ||
|
|
7096e47513 | ||
|
|
ce374ba877 | ||
|
|
0a19a49dfe | ||
|
|
443b0cad96 | ||
|
|
74843695eb | ||
|
|
0befb51327 | ||
|
|
dc31a72f50 |
@@ -51,11 +51,4 @@ jobs:
|
||||
USE_CUDA: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 2 --dist=loadfile -s ./tests/ | tee output.txt
|
||||
- name: cat output.txt
|
||||
run: cat output.txt
|
||||
- name: Upload output.txt
|
||||
uses: actions/upload-artifact@v1
|
||||
with:
|
||||
name: pytest_output
|
||||
path: output.txt
|
||||
python -m pytest -n 2 --dist=loadfile -s ./tests/
|
||||
|
||||
@@ -46,11 +46,4 @@ jobs:
|
||||
USE_CUDA: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ | tee output.txt
|
||||
- name: cat output.txt
|
||||
run: cat output.txt
|
||||
- name: Upload output.txt
|
||||
uses: actions/upload-artifact@v1
|
||||
with:
|
||||
name: pytest_output
|
||||
path: output.txt
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/
|
||||
|
||||
@@ -61,6 +61,13 @@ FlaubertForSequenceClassification
|
||||
:members:
|
||||
|
||||
|
||||
FlaubertForTokenClassification
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.FlaubertForTokenClassification
|
||||
:members:
|
||||
|
||||
|
||||
FlaubertForQuestionAnsweringSimple
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
@@ -114,4 +121,4 @@ TFFlaubertForQuestionAnsweringSimple
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TFFlaubertForQuestionAnsweringSimple
|
||||
:members:
|
||||
:members:
|
||||
|
||||
@@ -123,7 +123,7 @@ are 512 preceding tokens available to condition on).
|
||||
stride = 512
|
||||
|
||||
lls = []
|
||||
for i in tqdm(range(1, encodings.input_ids.size(1), stride)):
|
||||
for i in tqdm(range(0, encodings.input_ids.size(1), stride)):
|
||||
begin_loc = max(i + stride - max_length, 0)
|
||||
end_loc = i + stride
|
||||
input_ids = encodings.input_ids[:,begin_loc:end_loc].to(device)
|
||||
|
||||
@@ -108,11 +108,11 @@ any other model from the model hub):
|
||||
>>> model_name = "nlptown/bert-base-multilingual-uncased-sentiment"
|
||||
>>> model = AutoModelForSequenceClassification.from_pretrained(model_name)
|
||||
>>> tokenizer = AutoTokenizer.from_pretrained(model_name)
|
||||
>>> pipe = pipeline('sentiment-analysis', model=model, tokenizer=tokenizer)
|
||||
>>> classifier = pipeline('sentiment-analysis', model=model, tokenizer=tokenizer)
|
||||
>>> ## TENSORFLOW CODE
|
||||
>>> model_name = "nlptown/bert-base-multilingual-uncased-sentiment"
|
||||
>>> # This model only exists in PyTorch, so we use the `from_pt` flag to import that model in TensorFlow.
|
||||
>>> model = TFAutoModelForSequenceClassification.from_pretrained(model_name, from_pt=True)
|
||||
>>> model = TFAutoModelForSequenceClassification.from_pretrained(model_name, from_pt=True)
|
||||
>>> tokenizer = AutoTokenizer.from_pretrained(model_name)
|
||||
>>> classifier = pipeline('sentiment-analysis', model=model, tokenizer=tokenizer)
|
||||
|
||||
@@ -191,7 +191,7 @@ and get tensors back. You can specify all of that to the tokenizer:
|
||||
... return_tensors="tf"
|
||||
... )
|
||||
|
||||
The padding is automatically applied on the side the model expect it (in this case, on the right), with the
|
||||
The padding is automatically applied on the side expected by the model (in this case, on the right), with the
|
||||
padding token the model was pretrained with. The attention mask is also adapted to take the padding into account:
|
||||
|
||||
.. code-block::
|
||||
@@ -212,9 +212,9 @@ You can learn more about tokenizers :doc:`here <preprocessing>`.
|
||||
Using the model
|
||||
^^^^^^^^^^^^^^^
|
||||
|
||||
Once your input has been preprocessed by the tokenizer, you can directly send it to the model. As we mentioned, it will
|
||||
contain all the relevant information the model needs. If you're using a TensorFlow model, you can directly pass the
|
||||
dictionary keys to tensor, for a PyTorch model, you need to unpack the dictionary by adding :obj:`**`.
|
||||
Once your input has been preprocessed by the tokenizer, you can send it directly to the model. As we mentioned, it will
|
||||
contain all the relevant information the model needs. If you're using a TensorFlow model, you can pass the
|
||||
dictionary keys directly to tensor, for a PyTorch model, you need to unpack the dictionary by adding :obj:`**`.
|
||||
|
||||
.. code-block::
|
||||
|
||||
@@ -285,7 +285,7 @@ training loop. 🤗 Transformers also provides a :class:`~transformers.Trainer`
|
||||
you are using TensorFlow) class to help with your training (taking care of things such as distributed training, mixed
|
||||
precision, etc.). See the :doc:`training tutorial <training>` for more details.
|
||||
|
||||
Once your model is fine-tuned, you can save it with its tokenizer the following way:
|
||||
Once your model is fine-tuned, you can save it with its tokenizer in the following way:
|
||||
|
||||
::
|
||||
|
||||
@@ -329,7 +329,9 @@ pretrained model. Behind the scenes, the library has one model class per combina
|
||||
code is easy to access and tweak if you need to.
|
||||
|
||||
In our previous example, the model was called "distilbert-base-uncased-finetuned-sst-2-english", which means it's
|
||||
using the :doc:`DistilBERT </model_doc/distilbert>` architecture. The model automatically created is then a
|
||||
using the :doc:`DistilBERT </model_doc/distilbert>` architecture. As
|
||||
:class:`~transformers.AutoModelForSequenceClassification` (or :class:`~transformers.TFAutoModelForSequenceClassification`
|
||||
if you are using TensorFlow)` was used, the model automatically created is then a
|
||||
:class:`~transformers.DistilBertForSequenceClassification`. You can look at its documentation for all details relevant
|
||||
to that specific model, or browse the source code. This is how you would directly instantiate model and tokenizer
|
||||
without the auto magic:
|
||||
@@ -352,7 +354,7 @@ Customizing the model
|
||||
|
||||
If you want to change how the model itself is built, you can define your custom configuration class. Each architecture
|
||||
comes with its own relevant configuration (in the case of DistilBERT, :class:`~transformers.DistilBertConfig`) which
|
||||
allows you to specify any of the hidden dimension, dropout rate etc. If you do core modifications, like changing the
|
||||
allows you to specify any of the hidden dimension, dropout rate, etc. If you do core modifications, like changing the
|
||||
hidden size, you won't be able to use a pretrained model anymore and will need to train from scratch. You would then
|
||||
instantiate the model directly from this configuration.
|
||||
|
||||
|
||||
@@ -78,3 +78,32 @@ python examples/xla_spawn.py --num_cores 8 \
|
||||
```
|
||||
|
||||
Feedback and more use cases and benchmarks involving TPUs are welcome, please share with the community.
|
||||
|
||||
## Logging & Experiment tracking
|
||||
|
||||
You can easily log and monitor your runs code. [TensorBoard](https://www.tensorflow.org/tensorboard) and [Weights & Biases](https://docs.wandb.com/library/integrations/huggingface) are currently supported.
|
||||
|
||||
To use Weights & Biases, install the wandb package with:
|
||||
|
||||
```bash
|
||||
pip install wandb
|
||||
```
|
||||
|
||||
Then log in the command line:
|
||||
|
||||
```bash
|
||||
wandb login
|
||||
```
|
||||
|
||||
If you are in Jupyter or Colab, you should login with:
|
||||
|
||||
```python
|
||||
import wandb
|
||||
wandb.login()
|
||||
```
|
||||
|
||||
Whenever you use `Trainer` or `TFTrainer` classes, your losses, evaluation metrics, model topology and gradients (for `Trainer` only) will automatically be logged.
|
||||
|
||||
For advanced configuration and examples, refer to the [W&B documentation](https://docs.wandb.com/library/integrations/huggingface).
|
||||
|
||||
When using 🤗 Transformers with PyTorch Lightning, runs can be tracked through `WandbLogger`. Refer to related [documentation & examples](https://docs.wandb.com/library/frameworks/pytorch/lightning).
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
---
|
||||
language: hebrew
|
||||
tags:
|
||||
- pytorch
|
||||
- tf
|
||||
- gpt2
|
||||
- lm-head
|
||||
- causal-lm
|
||||
- pipeline:text-generation
|
||||
|
||||
thumbnail: https://avatars1.githubusercontent.com/u/3617152?norod.jpg
|
||||
widget:
|
||||
- text: "<|startoftext|>החוק השני של מועדון קרב הוא"
|
||||
- text: "<|startoftext|>ראש הממשלה בן גוריון"
|
||||
- text: "<|startoftext|>למידת מכונה (סרט)"
|
||||
- text: "<|startoftext|>מנשה פומפרניקל"
|
||||
- text: "<|startoftext|>אי שוויון "
|
||||
|
||||
license: mit
|
||||
---
|
||||
|
||||
|
||||
# hewiki-articles-distilGPT2py-il
|
||||
|
||||
## A tiny GPT2 model for generating Hebrew text
|
||||
|
||||
A distilGPT2 sized model. <br>
|
||||
Training data was hewiki-20200701-pages-articles-multistream.xml.bz2 from https://dumps.wikimedia.org/hewiki/20200701/ <br>
|
||||
XML has been converted to plain text using Wikipedia Extractor http://medialab.di.unipi.it/wiki/Wikipedia_Extractor <br>
|
||||
I then added <|startoftext|> and <|endoftext|> markers and deleted empty lines. <br>
|
||||
|
||||
#### How to use
|
||||
|
||||
```python
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
from transformers import GPT2Tokenizer, GPT2LMHeadModel
|
||||
|
||||
tokenizer = GPT2Tokenizer.from_pretrained("Norod78/hewiki-articles-distilGPT2py-il")
|
||||
model = GPT2LMHeadModel.from_pretrained("Norod78/hewiki-articles-distilGPT2py-il").eval()
|
||||
|
||||
bos_token = tokenizer.bos_token #Beginning of sentace
|
||||
eos_token = tokenizer.eos_token #End of sentence
|
||||
|
||||
def generate_word(model, tokens_tensor, temperature=1.0):
|
||||
"""
|
||||
Sample a word given a tensor of tokens of previous words from a model. Given
|
||||
the words we have, sample a plausible word. Temperature is used for
|
||||
controlling randomness. If using temperature==0 we simply use a greedy arg max.
|
||||
Else, we sample from a multinomial distribution using a lower inverse
|
||||
temperature to allow for more randomness to escape repetitions.
|
||||
"""
|
||||
with torch.no_grad():
|
||||
outputs = model(tokens_tensor)
|
||||
predictions = outputs[0]
|
||||
if temperature>0:
|
||||
# Make the distribution more or less skewed based on the temperature
|
||||
predictions = outputs[0]/temperature
|
||||
# Sample from the distribution
|
||||
softmax = nn.Softmax(dim=0)
|
||||
predicted_index = torch.multinomial(softmax(predictions[0,-1,:]),1).item()
|
||||
# Simply take the arg-max of the distribution
|
||||
else:
|
||||
predicted_index = torch.argmax(predictions[0, -1, :]).item()
|
||||
# Decode the encoding to the corresponding word
|
||||
predicted_text = tokenizer.decode([predicted_index])
|
||||
return predicted_text
|
||||
|
||||
def generate_sentence(model, tokenizer, initial_text, temperature=1.0):
|
||||
""" Generate a sentence given some initial text using a model and a tokenizer.
|
||||
Returns the new sentence. """
|
||||
|
||||
# Encode a text inputs
|
||||
text = ""
|
||||
sentence = text
|
||||
|
||||
# We avoid an infinite loop by setting a maximum range
|
||||
for i in range(0,84):
|
||||
indexed_tokens = tokenizer.encode(initial_text + text)
|
||||
|
||||
# Convert indexed tokens in a PyTorch tensor
|
||||
tokens_tensor = torch.tensor([indexed_tokens])
|
||||
|
||||
new_word = generate_word(model, tokens_tensor, temperature=temperature)
|
||||
|
||||
# Here the temperature is slowly decreased with each generated word,
|
||||
# this ensures that the sentence (ending) makes more sense.
|
||||
# We don't decrease to a temperature of 0.0 to leave some randomness in.
|
||||
if temperature<(1-0.008):
|
||||
temperature += 0.008
|
||||
else:
|
||||
temperature = 0.996
|
||||
|
||||
text = text+new_word
|
||||
|
||||
# Stop generating new words when we have reached the end of the line or the poem
|
||||
if eos_token in new_word:
|
||||
# returns new sentence and whether poem is done
|
||||
return (text.replace(eos_token,"").strip(), True)
|
||||
elif '/' in new_word:
|
||||
return (text.strip(), False)
|
||||
elif bos_token in new_word:
|
||||
return (text.replace(bos_token,"").strip(), False)
|
||||
|
||||
return (text, True)
|
||||
|
||||
for output_num in range(1,5):
|
||||
init_text = "בוקר טוב"
|
||||
text = bos_token + init_text
|
||||
for i in range(0,84):
|
||||
sentence = generate_sentence(model, tokenizer, text, temperature=0.9)
|
||||
text = init_text + sentence[0]
|
||||
print(text)
|
||||
if (sentence[1] == True):
|
||||
break
|
||||
```
|
||||
@@ -21,17 +21,17 @@ By [Mawdoo3-AI](https://ai.mawdoo3.com/).
|
||||
Instead of training the Multi-dialect Arabic BERT model from scratch, we initialized the weights of the model using [Arabic-BERT](https://github.com/alisafaya/Arabic-BERT) and trained it on 10M arabic tweets from the unlabled data of [The Nuanced Arabic Dialect Identification (NADI) shared task](https://sites.google.com/view/nadi-shared-task).
|
||||
|
||||
### To cite this work
|
||||
Please cite this paper for now:
|
||||
|
||||
```
|
||||
@inproceedings{talafha-etal-2020-nadi,
|
||||
title ={{Multi-dialect Arabic BERT for Country-level Dialect Identification}},
|
||||
author = {Talafha, Bashar, Ali, Mohammad, Za'ter, Muhy Eddin, Seelawi, Haitham, Tuffaha, Ibraheem, Samir, Mostafa, Farhan, Wael and Al-Natsheh, Hussein},
|
||||
booktitle ={{Proceedings of the Fifth Arabic Natural Language Processing Workshop (WANLP2020)}},
|
||||
year = {2020},
|
||||
address = {Barcelona, Spain}
|
||||
@misc{talafha2020multidialect,
|
||||
title={Multi-Dialect Arabic BERT for Country-Level Dialect Identification},
|
||||
author={Bashar Talafha and Mohammad Ali and Muhy Eddin Za'ter and Haitham Seelawi and Ibraheem Tuffaha and Mostafa Samir and Wael Farhan and Hussein T. Al-Natsheh},
|
||||
year={2020},
|
||||
eprint={2007.05612},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL}
|
||||
}
|
||||
```
|
||||
We will update the BibTeX once the paper published.
|
||||
|
||||
### Usage
|
||||
The model weights can be loaded using `transformers` library by HuggingFace.
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
## RuDR-BERT
|
||||
|
||||
RuDR-BERT - Multilingual, Cased, which pretrained on the raw part of the RuDReC corpus (1.4M reviews). Pre-training was based on the [original BERT code](https://github.com/google-research/bert) provided by Google. In particular, Multi-BERT was for used for initialization; vocabulary of Russian subtokens and parameters are the same as in Multi-BERT. Training details are described in our paper. \
|
||||
link: https://yadi.sk/d/-PTn0xhk1PqvgQ
|
||||
|
||||
|
||||
## Citing & Authors
|
||||
|
||||
If you find this repository helpful, feel free to cite our publication:
|
||||
|
||||
[1] https://arxiv.org/abs/2004.03659
|
||||
```
|
||||
@misc{tutubalina2020russian,
|
||||
title={The Russian Drug Reaction Corpus and Neural Models for Drug Reactions and Effectiveness Detection in User Reviews},
|
||||
author={Elena Tutubalina and Ilseyar Alimova and Zulfat Miftahutdinov and Andrey Sakhovskiy and Valentin Malykh and Sergey Nikolenko},
|
||||
year={2020},
|
||||
eprint={2004.03659},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL}
|
||||
}
|
||||
```
|
||||
[2] Tutubalina, EV and Miftahutdinov, Z Sh and Nugmanov, RI and Madzhidov, TI and Nikolenko, SI and Alimova, IS and Tropsha, AE Using semantic analysis of texts for the identification of drugs with similar therapeutic effects.
|
||||
[link to paper](https://www.researchgate.net/profile/Elena_Tutubalina/publication/323751823_Using_semantic_analysis_of_texts_for_the_identification_of_drugs_with_similar_therapeutic_effects/links/5bf7cfc3299bf1a0202cbc1f/Using-semantic-analysis-of-texts-for-the-identification-of-drugs-with-similar-therapeutic-effects.pdf)
|
||||
```
|
||||
@article{tutubalina2017using,
|
||||
title={Using semantic analysis of texts for the identification of drugs with similar therapeutic effects},
|
||||
author={Tutubalina, EV and Miftahutdinov, Z Sh and Nugmanov, RI and Madzhidov, TI and Nikolenko, SI and Alimova, IS and Tropsha, AE},
|
||||
journal={Russian Chemical Bulletin},
|
||||
volume={66},
|
||||
number={11},
|
||||
pages={2180--2189},
|
||||
year={2017},
|
||||
publisher={Springer}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,90 @@
|
||||
## About
|
||||
|
||||
The *french-postag-model* is a part of speech tagging model for French that was trained on the *free-french-treebank* dataset available on
|
||||
[github](https://github.com/nicolashernandez/free-french-treebank). The base tokenizer and model used for training is *'bert-base-multilingual-cased'*.
|
||||
|
||||
## Supported Tags
|
||||
|
||||
It uses the following tags:
|
||||
|
||||
| Tag | Category | Extra Info |
|
||||
|----------|:------------------------------:|------------:|
|
||||
| ADJ | adjectif | |
|
||||
| ADJWH | adjectif | |
|
||||
| ADV | adverbe | |
|
||||
| ADVWH | adverbe | |
|
||||
| CC | conjonction de coordination | |
|
||||
| CLO | pronom | obj |
|
||||
| CLR | pronom | refl |
|
||||
| CLS | pronom | suj |
|
||||
| CS | conjonction de subordination | |
|
||||
| DET | déterminant | |
|
||||
| DETWH | déterminant | |
|
||||
| ET | mot étranger | |
|
||||
| I | interjection | |
|
||||
| NC | nom commun | |
|
||||
| NPP | nom propre | |
|
||||
| P | préposition | |
|
||||
| P+D | préposition + déterminant | |
|
||||
| PONCT | signe de ponctuation | |
|
||||
| PREF | préfixe | |
|
||||
| PRO | autres pronoms | |
|
||||
| PROREL | autres pronoms | rel |
|
||||
| PROWH | autres pronoms | int |
|
||||
| U | ? | |
|
||||
| V | verbe | |
|
||||
| VIMP | verbe imperatif | |
|
||||
| VINF | verbe infinitif | |
|
||||
| VPP | participe passé | |
|
||||
| VPR | participe présent | |
|
||||
| VS | subjonctif | |
|
||||
|
||||
More information on the tags can be found here:
|
||||
|
||||
http://alpage.inria.fr/statgram/frdep/Publications/crabbecandi-taln2008-final.pdf
|
||||
|
||||
## Usage
|
||||
|
||||
The usage of this model follows the common transformers patterns. Here is a short example of its usage:
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModelForTokenClassification
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("gilf/french-postag-model")
|
||||
model = AutoModelForTokenClassification.from_pretrained("gilf/french-postag-model")
|
||||
|
||||
from transformers import pipeline
|
||||
|
||||
nlp_token_class = pipeline('ner', model=model, tokenizer=tokenizer, grouped_entities=True)
|
||||
|
||||
nlp_token_class('Face à un choc inédit, les mesures mises en place par le gouvernement ont permis une protection forte et efficace des ménages')
|
||||
```
|
||||
|
||||
The lines above would display something like this on a Jupyter notebook:
|
||||
|
||||
```
|
||||
[{'entity_group': 'PONCT', 'score': 0.0742340236902237, 'word': '[CLS]'},
|
||||
{'entity_group': 'U', 'score': 0.9995399713516235, 'word': 'Face'},
|
||||
{'entity_group': 'P', 'score': 0.9999609589576721, 'word': 'à'},
|
||||
{'entity_group': 'DET', 'score': 0.9999597072601318, 'word': 'un'},
|
||||
{'entity_group': 'NC', 'score': 0.9998948276042938, 'word': 'choc'},
|
||||
{'entity_group': 'ADJ', 'score': 0.995318204164505, 'word': 'inédit'},
|
||||
{'entity_group': 'PONCT', 'score': 0.9999793171882629, 'word': ','},
|
||||
{'entity_group': 'DET', 'score': 0.999964714050293, 'word': 'les'},
|
||||
{'entity_group': 'NC', 'score': 0.999936580657959, 'word': 'mesures'},
|
||||
{'entity_group': 'VPP', 'score': 0.9995776414871216, 'word': 'mises'},
|
||||
{'entity_group': 'P', 'score': 0.99996417760849, 'word': 'en'},
|
||||
{'entity_group': 'NC', 'score': 0.999882161617279, 'word': 'place'},
|
||||
{'entity_group': 'P', 'score': 0.9999671578407288, 'word': 'par'},
|
||||
{'entity_group': 'DET', 'score': 0.9999637603759766, 'word': 'le'},
|
||||
{'entity_group': 'NC', 'score': 0.9999350309371948, 'word': 'gouvernement'},
|
||||
{'entity_group': 'V', 'score': 0.9999298453330994, 'word': 'ont'},
|
||||
{'entity_group': 'VPP', 'score': 0.9998740553855896, 'word': 'permis'},
|
||||
{'entity_group': 'DET', 'score': 0.9999625086784363, 'word': 'une'},
|
||||
{'entity_group': 'NC', 'score': 0.9999420046806335, 'word': 'protection'},
|
||||
{'entity_group': 'ADJ', 'score': 0.9998913407325745, 'word': 'forte'},
|
||||
{'entity_group': 'CC', 'score': 0.9998615980148315, 'word': 'et'},
|
||||
{'entity_group': 'ADJ', 'score': 0.9998483657836914, 'word': 'efficace'},
|
||||
{'entity_group': 'P+D', 'score': 0.9987645149230957, 'word': 'des'},
|
||||
{'entity_group': 'NC', 'score': 0.8720395267009735, 'word': 'ménages [SEP]'}]
|
||||
```
|
||||
@@ -0,0 +1,46 @@
|
||||
## CodeBERT-base-mlm
|
||||
Pretrained weights for [CodeBERT: A Pre-Trained Model for Programming and Natural Languages](https://arxiv.org/abs/2002.08155).
|
||||
|
||||
### Training Data
|
||||
The model is trained on the code corpus of [CodeSearchNet](https://github.com/github/CodeSearchNet)
|
||||
|
||||
### Training Objective
|
||||
This model is initialized with Roberta-base and trained with a simple MLM (Masked Language Model) objective.
|
||||
|
||||
### Usage
|
||||
```python
|
||||
from transformers import RobertaTokenizer, RobertaForMaskedLM, pipeline
|
||||
|
||||
model = RobertaForMaskedLM.from_pretrained('microsoft/codebert-base-mlm')
|
||||
tokenizer = RobertaTokenizer.from_pretrained('microsoft/codebert-base-mlm')
|
||||
|
||||
code_example = "if (x is not None) <mask> (x>1)"
|
||||
fill_mask = pipeline('fill-mask', model=model, tokenizer=tokenizer)
|
||||
|
||||
outputs = fill_mask(code_example)
|
||||
print(outputs)
|
||||
```
|
||||
Expected results:
|
||||
```
|
||||
{'sequence': '<s> if (x is not None) and (x>1)</s>', 'score': 0.6049249172210693, 'token': 8}
|
||||
{'sequence': '<s> if (x is not None) or (x>1)</s>', 'score': 0.30680200457572937, 'token': 50}
|
||||
{'sequence': '<s> if (x is not None) if (x>1)</s>', 'score': 0.02133703976869583, 'token': 114}
|
||||
{'sequence': '<s> if (x is not None) then (x>1)</s>', 'score': 0.018607674166560173, 'token': 172}
|
||||
{'sequence': '<s> if (x is not None) AND (x>1)</s>', 'score': 0.007619690150022507, 'token': 4248}
|
||||
```
|
||||
|
||||
### Reference
|
||||
1. [Bimodal CodeBERT trained with MLM+RTD objective](https://huggingface.co/microsoft/codebert-base) (suitable for code search and document generation)
|
||||
2. 🤗 [Hugging Face's CodeBERTa](https://huggingface.co/huggingface/CodeBERTa-small-v1) (small size, 6 layers)
|
||||
|
||||
### Citation
|
||||
```bibtex
|
||||
@misc{feng2020codebert,
|
||||
title={CodeBERT: A Pre-Trained Model for Programming and Natural Languages},
|
||||
author={Zhangyin Feng and Daya Guo and Duyu Tang and Nan Duan and Xiaocheng Feng and Ming Gong and Linjun Shou and Bing Qin and Ting Liu and Daxin Jiang and Ming Zhou},
|
||||
year={2020},
|
||||
eprint={2002.08155},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,27 @@
|
||||
## CodeBERT-base
|
||||
Pretrained weights for [CodeBERT: A Pre-Trained Model for Programming and Natural Languages](https://arxiv.org/abs/2002.08155).
|
||||
|
||||
### Training Data
|
||||
The model is trained on bi-modal data (documents & code) of [CodeSearchNet](https://github.com/github/CodeSearchNet)
|
||||
|
||||
### Training Objective
|
||||
This model is initialized with Roberta-base and trained with MLM+RTD objective (cf. the paper).
|
||||
|
||||
### Usage
|
||||
Please see [the official repository](https://github.com/microsoft/CodeBERT) for scripts that support "code search" and "code-to-document generation".
|
||||
|
||||
### Reference
|
||||
1. [CodeBERT trained with Masked LM objective](https://huggingface.co/microsoft/codebert-base-mlm) (suitable for code completion)
|
||||
2. 🤗 [Hugging Face's CodeBERTa](https://huggingface.co/huggingface/CodeBERTa-small-v1) (small size, 6 layers)
|
||||
|
||||
### Citation
|
||||
```bibtex
|
||||
@misc{feng2020codebert,
|
||||
title={CodeBERT: A Pre-Trained Model for Programming and Natural Languages},
|
||||
author={Zhangyin Feng and Daya Guo and Duyu Tang and Nan Duan and Xiaocheng Feng and Ming Gong and Linjun Shou and Bing Qin and Ting Liu and Daxin Jiang and Ming Zhou},
|
||||
year={2020},
|
||||
eprint={2002.08155},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL}
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,3 @@
|
||||
---
|
||||
language: spanish
|
||||
---
|
||||
@@ -0,0 +1,177 @@
|
||||
---
|
||||
language: Portuguese
|
||||
---
|
||||
|
||||
# GPorTuguese-2: a Language Model for Portuguese text generation (and more NLP tasks...)
|
||||
|
||||
## Introduction
|
||||
|
||||
GPorTuguese-2 (Portuguese GPT-2 small) is a state-of-the-art language model for Portuguese based on the GPT-2 small model.
|
||||
|
||||
It was trained on Portuguese Wikipedia using **Transfer Learning and Fine-tuning techniques** in just over a day, on one GPU NVIDIA V100 32GB and with a little more than 1GB of training data.
|
||||
|
||||
It is a proof-of-concept that it is possible to get a state-of-the-art language model in any language with low ressources.
|
||||
|
||||
It was fine-tuned from the [English pre-trained GPT-2 small](https://huggingface.co/gpt2) using the Hugging Face libraries (Transformers and Tokenizers) wrapped into the [fastai v2](https://dev.fast.ai/) Deep Learning framework. All the fine-tuning fastai v2 techniques were used.
|
||||
|
||||
It is now available on Hugging Face. For further information or requests, please go to "[Faster than training from scratch — Fine-tuning the English GPT-2 in any language with Hugging Face and fastai v2 (practical case with Portuguese)](https://medium.com/@pierre_guillou/faster-than-training-from-scratch-fine-tuning-the-english-gpt-2-in-any-language-with-hugging-f2ec05c98787)".
|
||||
|
||||
## Model
|
||||
|
||||
| Model | #params | Model file (pt/tf) | Arch. | Training /Validation data (text) |
|
||||
|-------------------------|---------|--------------------|-------------|------------------------------------------|
|
||||
| `gpt2-small-portuguese` | 124M | 487M / 475M | GPT-2 small | Portuguese Wikipedia (1.28 GB / 0.32 GB) |
|
||||
|
||||
## Evaluation results
|
||||
In a little more than a day (we only used one GPU NVIDIA V100 32GB; through a Distributed Data Parallel (DDP) training mode, we could have divided by three this time to 10 hours, just with 2 GPUs), we got a loss of 3.17, an **accuracy of 37.99%** and a **perplexity of 23.76** (see the validation results table below).
|
||||
|
||||
| after ... epochs | loss | accuracy (%) | perplexity | time by epoch | cumulative time |
|
||||
|------------------|------|--------------|------------|---------------|-----------------|
|
||||
| 0 | 9.95 | 9.90 | 20950.94 | 00:00:00 | 00:00:00 |
|
||||
| 1 | 3.64 | 32.52 | 38.12 | 5:48:31 | 5:48:31 |
|
||||
| 2 | 3.30 | 36.29 | 27.16 | 5:38:18 | 11:26:49 |
|
||||
| 3 | 3.21 | 37.46 | 24.71 | 6:20:51 | 17:47:40 |
|
||||
| 4 | 3.19 | 37.74 | 24.21 | 6:06:29 | 23:54:09 |
|
||||
| 5 | 3.17 | 37.99 | 23.76 | 6:16:22 | 30:10:31 |
|
||||
|
||||
## GPT-2
|
||||
|
||||
*Note: information copied/pasted from [Model: gpt2 >> GPT-2](https://huggingface.co/gpt2#gpt-2)*
|
||||
|
||||
Pretrained model on English language using a causal language modeling (CLM) objective. It was introduced in this [paper](https://d4mucfpksywv.cloudfront.net/better-language-models/language_models_are_unsupervised_multitask_learners.pdf) and first released at this [page](https://openai.com/blog/better-language-models/) (February 14, 2019).
|
||||
|
||||
Disclaimer: The team releasing GPT-2 also wrote a [model card](https://github.com/openai/gpt-2/blob/master/model_card.md) for their model. Content from this model card has been written by the Hugging Face team to complete the information they provided and give specific examples of bias.
|
||||
|
||||
## Model description
|
||||
|
||||
*Note: information copied/pasted from [Model: gpt2 >> Model description](https://huggingface.co/gpt2#model-description)*
|
||||
|
||||
GPT-2 is a transformers model pretrained on a very large corpus of English data in a self-supervised fashion. This means it was pretrained on the raw texts only, with no humans labelling them in any way (which is why it can use lots of publicly available data) with an automatic process to generate inputs and labels from those texts. More precisely, it was trained to guess the next word in sentences.
|
||||
|
||||
More precisely, inputs are sequences of continuous text of a certain length and the targets are the same sequence, shifted one token (word or piece of word) to the right. The model uses internally a mask-mechanism to make sure the predictions for the token `i` only uses the inputs from `1` to `i` but not the future tokens.
|
||||
|
||||
This way, the model learns an inner representation of the English language that can then be used to extract features useful for downstream tasks. The model is best at what it was pretrained for however, which is generating texts from a prompt.
|
||||
|
||||
## How to use GPorTuguese-2 with HuggingFace (PyTorch)
|
||||
|
||||
The following code use PyTorch. To use TensorFlow, check the below corresponding paragraph.
|
||||
|
||||
### Load GPorTuguese-2 and its sub-word tokenizer (Byte-level BPE)
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModelWithLMHead
|
||||
import torch
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("pierreguillou/gpt2-small-portuguese")
|
||||
model = AutoModelWithLMHead.from_pretrained("pierreguillou/gpt2-small-portuguese")
|
||||
|
||||
# Get sequence length max of 1024
|
||||
tokenizer.model_max_length=1024
|
||||
|
||||
model.eval() # disable dropout (or leave in train mode to finetune)
|
||||
```
|
||||
|
||||
### Generate one word
|
||||
|
||||
```python
|
||||
# input sequence
|
||||
text = "Quem era Jim Henson? Jim Henson era um"
|
||||
inputs = tokenizer(text, return_tensors="pt")
|
||||
|
||||
# model output
|
||||
outputs = model(**inputs, labels=inputs["input_ids"])
|
||||
loss, logits = outputs[:2]
|
||||
predicted_index = torch.argmax(logits[0, -1, :]).item()
|
||||
predicted_text = tokenizer.decode([predicted_index])
|
||||
|
||||
# results
|
||||
print('input text:', text)
|
||||
print('predicted text:', predicted_text)
|
||||
|
||||
# input text: Quem era Jim Henson? Jim Henson era um
|
||||
# predicted text: homem
|
||||
```
|
||||
|
||||
### Generate one full sequence
|
||||
|
||||
```python
|
||||
# input sequence
|
||||
text = "Quem era Jim Henson? Jim Henson era um"
|
||||
inputs = tokenizer(text, return_tensors="pt")
|
||||
|
||||
# model output using Top-k sampling text generation method
|
||||
sample_outputs = model.generate(inputs.input_ids,
|
||||
pad_token_id=50256,
|
||||
do_sample=True,
|
||||
max_length=50, # put the token number you want
|
||||
top_k=40,
|
||||
num_return_sequences=1)
|
||||
|
||||
# generated sequence
|
||||
for i, sample_output in enumerate(sample_outputs):
|
||||
print(">> Generated text {}\n\n{}".format(i+1, tokenizer.decode(sample_output.tolist())))
|
||||
|
||||
# >> Generated text
|
||||
# Quem era Jim Henson? Jim Henson era um executivo de televisão e diretor de um grande estúdio de cinema mudo chamado Selig,
|
||||
# depois que o diretor de cinema mudo Georges Seuray dirigiu vários filmes para a Columbia e o estúdio.
|
||||
```
|
||||
|
||||
## How to use GPorTuguese-2 with HuggingFace (TensorFlow)
|
||||
|
||||
The following code use TensorFlow. To use PyTorch, check the above corresponding paragraph.
|
||||
|
||||
### Load GPorTuguese-2 and its sub-word tokenizer (Byte-level BPE)
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, TFAutoModelWithLMHead
|
||||
import tensorflow as tf
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("pierreguillou/gpt2-small-portuguese")
|
||||
model = TFAutoModelWithLMHead.from_pretrained("pierreguillou/gpt2-small-portuguese")
|
||||
|
||||
# Get sequence length max of 1024
|
||||
tokenizer.model_max_length=1024
|
||||
|
||||
model.eval() # disable dropout (or leave in train mode to finetune)
|
||||
```
|
||||
|
||||
### Generate one full sequence
|
||||
|
||||
```python
|
||||
# input sequence
|
||||
text = "Quem era Jim Henson? Jim Henson era um"
|
||||
inputs = tokenizer.encode(text, return_tensors="tf")
|
||||
|
||||
# model output using Top-k sampling text generation method
|
||||
outputs = model.generate(inputs, eos_token_id=50256, pad_token_id=50256,
|
||||
do_sample=True,
|
||||
max_length=40,
|
||||
top_k=40)
|
||||
print(tokenizer.decode(outputs[0]))
|
||||
|
||||
# >> Generated text
|
||||
# Quem era Jim Henson? Jim Henson era um amigo familiar da família. Ele foi contratado pelo seu pai
|
||||
# para trabalhar como aprendiz no escritório de um escritório de impressão, e então começou a ganhar dinheiro
|
||||
|
||||
```
|
||||
|
||||
## Limitations and bias
|
||||
|
||||
The training data used for this model come from Portuguese Wikipedia. We know it contains a lot of unfiltered content from the internet, which is far from neutral. As the openAI team themselves point out in their model card:
|
||||
|
||||
> Because large-scale language models like GPT-2 do not distinguish fact from fiction, we don’t support use-cases that require the generated text to be true. Additionally, language models like GPT-2 reflect the biases inherent to the systems they were trained on, so we do not recommend that they be deployed into systems that interact with humans > unless the deployers first carry out a study of biases relevant to the intended use-case. We found no statistically significant difference in gender, race, and religious bias probes between 774M and 1.5B, implying all versions of GPT-2 should be approached with similar levels of caution around use cases that are sensitive to biases around human attributes.
|
||||
|
||||
## Author
|
||||
|
||||
Portuguese GPT-2 small was trained and evaluated by [Pierre GUILLOU](https://www.linkedin.com/in/pierreguillou/) thanks to the computing power of the GPU (GPU NVIDIA V100 32 Go) of the [AI Lab](https://www.linkedin.com/company/ailab-unb/) (University of Brasilia) to which I am attached as an Associate Researcher in NLP and the participation of its directors in the definition of NLP strategy, Professors Fabricio Ataides Braz and Nilton Correia da Silva.
|
||||
|
||||
## Citation
|
||||
If you use our work, please cite:
|
||||
|
||||
```bibtex
|
||||
@inproceedings{pierre2020gpt2smallportuguese,
|
||||
title={GPorTuguese-2 (Portuguese GPT-2 small): a Language Model for Portuguese text generation (and more NLP tasks...)},
|
||||
author={Pierre Guillou},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
@@ -353,6 +353,7 @@ if is_torch_available():
|
||||
FlaubertModel,
|
||||
FlaubertWithLMHeadModel,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
FlaubertForQuestionAnswering,
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
@@ -377,6 +378,7 @@ if is_torch_available():
|
||||
ReformerModel,
|
||||
ReformerForMaskedLM,
|
||||
ReformerModelWithLMHead,
|
||||
ReformerForSequenceClassification,
|
||||
ReformerForQuestionAnswering,
|
||||
REFORMER_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
)
|
||||
|
||||
@@ -17,8 +17,8 @@ USE_AMP = False
|
||||
|
||||
def train_command_factory(args: Namespace):
|
||||
"""
|
||||
Factory function used to instantiate serving server from provided command line arguments.
|
||||
:return: ServeCommand
|
||||
Factory function used to instantiate training command from provided command line arguments.
|
||||
:return: TrainCommand
|
||||
"""
|
||||
return TrainCommand(args)
|
||||
|
||||
|
||||
@@ -100,6 +100,7 @@ from .modeling_encoder_decoder import EncoderDecoderModel
|
||||
from .modeling_flaubert import (
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
FlaubertModel,
|
||||
FlaubertWithLMHeadModel,
|
||||
)
|
||||
@@ -326,6 +327,7 @@ MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING = OrderedDict(
|
||||
[
|
||||
(DistilBertConfig, DistilBertForTokenClassification),
|
||||
(CamembertConfig, CamembertForTokenClassification),
|
||||
(FlaubertConfig, FlaubertForTokenClassification),
|
||||
(XLMConfig, XLMForTokenClassification),
|
||||
(XLMRobertaConfig, XLMRobertaForTokenClassification),
|
||||
(LongformerConfig, LongformerForTokenClassification),
|
||||
@@ -1552,6 +1554,7 @@ class AutoModelForTokenClassification:
|
||||
- isInstance of `bert` configuration class: :class:`~transformers.BertModelForTokenClassification` (Bert model)
|
||||
- isInstance of `albert` configuration class: :class:`~transformers.AlbertForTokenClassification` (AlBert model)
|
||||
- isInstance of `xlnet` configuration class: :class:`~transformers.XLNetModelForTokenClassification` (XLNet model)
|
||||
- isInstance of `flaubert` configuration class: :class:`~transformers.FlaubertForTokenClassification` (Flaubert model)
|
||||
- isInstance of `camembert` configuration class: :class:`~transformers.CamembertModelForTokenClassification` (Camembert model)
|
||||
- isInstance of `roberta` configuration class: :class:`~transformers.RobertaModelForTokenClassification` (Roberta model)
|
||||
- isInstance of `electra` configuration class: :class:`~transformers.ElectraForTokenClassification` (Electra model)
|
||||
@@ -1589,6 +1592,7 @@ class AutoModelForTokenClassification:
|
||||
- `camembert`: :class:`~transformers.CamembertForTokenClassification` (Camembert model)
|
||||
- `bert`: :class:`~transformers.BertForTokenClassification` (Bert model)
|
||||
- `xlnet`: :class:`~transformers.XLNetForTokenClassification` (XLNet model)
|
||||
- `flaubert`: :class:`~transformers.FlaubertForTokenClassification` (Flaubert model)
|
||||
- `roberta`: :class:`~transformers.RobertaForTokenClassification` (Roberta model)
|
||||
- `electra`: :class:`~transformers.ElectraForTokenClassification` (Electra model)
|
||||
|
||||
|
||||
@@ -628,8 +628,8 @@ class SelfAttention(nn.Module):
|
||||
self.out_proj = nn.Linear(embed_dim, embed_dim, bias=bias)
|
||||
self.cache_key = "encoder_decoder" if self.encoder_decoder_attention else "self"
|
||||
|
||||
def _shape(self, tensor, dim_0, bsz):
|
||||
return tensor.contiguous().view(dim_0, bsz * self.num_heads, self.head_dim).transpose(0, 1)
|
||||
def _shape(self, tensor, seq_len, bsz):
|
||||
return tensor.contiguous().view(seq_len, bsz * self.num_heads, self.head_dim).transpose(0, 1)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
@@ -648,10 +648,9 @@ class SelfAttention(nn.Module):
|
||||
# get here for encoder decoder cause of static_kv
|
||||
if layer_state is not None: # reuse k,v and encoder_padding_mask
|
||||
saved_state = layer_state.get(self.cache_key, {})
|
||||
if "prev_key" in saved_state:
|
||||
if "prev_key" in saved_state and static_kv:
|
||||
# previous time steps are cached - no need to recompute key and value if they are static
|
||||
if static_kv:
|
||||
key = None
|
||||
key = None
|
||||
else:
|
||||
saved_state = None
|
||||
layer_state = {}
|
||||
@@ -738,37 +737,14 @@ class SelfAttention(nn.Module):
|
||||
v = torch.cat([prev_value, v], dim=1)
|
||||
assert k is not None and v is not None
|
||||
prev_key_padding_mask: Optional[Tensor] = saved_state.get("prev_key_padding_mask", None)
|
||||
key_padding_mask = self._cat_prev_key_padding_mask(
|
||||
key_padding_mask, prev_key_padding_mask, bsz, k.size(1), static_kv
|
||||
)
|
||||
return k, v, key_padding_mask
|
||||
|
||||
@staticmethod
|
||||
def _cat_prev_key_padding_mask(
|
||||
key_padding_mask: Optional[Tensor],
|
||||
prev_key_padding_mask: Optional[Tensor],
|
||||
batch_size: int,
|
||||
src_len: int,
|
||||
static_kv: bool,
|
||||
) -> Optional[Tensor]:
|
||||
# saved key padding masks have shape (bsz, seq_len)
|
||||
if prev_key_padding_mask is not None:
|
||||
if static_kv:
|
||||
new_key_padding_mask = prev_key_padding_mask
|
||||
else:
|
||||
new_key_padding_mask = torch.cat([prev_key_padding_mask, key_padding_mask], dim=1)
|
||||
|
||||
elif key_padding_mask is not None:
|
||||
filler = torch.zeros(
|
||||
batch_size,
|
||||
src_len - key_padding_mask.size(1),
|
||||
dtype=key_padding_mask.dtype,
|
||||
device=key_padding_mask.device,
|
||||
)
|
||||
new_key_padding_mask = torch.cat([filler, key_padding_mask], dim=1)
|
||||
else:
|
||||
new_key_padding_mask = prev_key_padding_mask
|
||||
return new_key_padding_mask
|
||||
new_key_padding_mask = key_padding_mask
|
||||
return k, v, new_key_padding_mask
|
||||
|
||||
|
||||
class BartClassificationHead(nn.Module):
|
||||
|
||||
@@ -28,6 +28,7 @@ from .modeling_xlm import (
|
||||
XLMForQuestionAnswering,
|
||||
XLMForQuestionAnsweringSimple,
|
||||
XLMForSequenceClassification,
|
||||
XLMForTokenClassification,
|
||||
XLMModel,
|
||||
XLMWithLMHeadModel,
|
||||
get_masks,
|
||||
@@ -326,6 +327,25 @@ class FlaubertForSequenceClassification(XLMForSequenceClassification):
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
"""Flaubert Model with a token classification head on top (a linear layer on top of
|
||||
the hidden-states output) e.g. for Named-Entity-Recognition (NER) tasks. """,
|
||||
FLAUBERT_START_DOCSTRING,
|
||||
)
|
||||
class FlaubertForTokenClassification(XLMForTokenClassification):
|
||||
"""
|
||||
This class overrides :class:`~transformers.XLMForTokenClassification`. Please check the
|
||||
superclass for the appropriate documentation alongside usage examples.
|
||||
"""
|
||||
|
||||
config_class = FlaubertConfig
|
||||
|
||||
def __init__(self, config):
|
||||
super().__init__(config)
|
||||
self.transformer = FlaubertModel(config)
|
||||
self.init_weights()
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
"""Flaubert Model with a span classification head on top for extractive question-answering tasks like SQuAD (a linear layers on top of
|
||||
the hidden-states output to compute `span start logits` and `span end logits`). """,
|
||||
|
||||
@@ -442,12 +442,14 @@ class LongformerSelfAttention(nn.Module):
|
||||
if output_attentions:
|
||||
if is_global_attn:
|
||||
# With global attention, return global attention probabilities only
|
||||
# batch_size x num_heads x max_num_global_attention_tokens x sequence_length
|
||||
# which is the attention weights from tokens with global attention to all tokens
|
||||
# It doesn't not return local attention
|
||||
# In case of variable number of global attantion in the rows of a batch,
|
||||
# attn_probs are padded with -10000.0 attention scores
|
||||
attn_probs = attn_probs.view(batch_size, self.num_heads, max_num_global_attn_indices, seq_len)
|
||||
# batch_size x num_heads x sequence_length x window_size
|
||||
# which is the attention weights from all tokens to all tokens for global attention
|
||||
# It doesn't not return local attention. Only tokens with global attention have values > 0.0
|
||||
attn_probs = attn_probs[:, :, :, :max_num_global_attn_indices]
|
||||
# pad attn_probs to max length with 0.0 since global attn did not attend there
|
||||
window_size = self.one_sided_attn_window_size * 2 + 1
|
||||
attn_probs = F.pad(attn_probs, (0, window_size - max_num_global_attn_indices), value=0.0,)
|
||||
attn_probs = attn_probs.permute(0, 2, 1, 3)
|
||||
else:
|
||||
# without global attention, return local attention probabilities
|
||||
# batch_size x num_heads x sequence_length x window_size
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -600,7 +600,7 @@ class XLNetModelOutput(ModelOutput):
|
||||
@dataclass
|
||||
class XLNetLMHeadModelOutput(ModelOutput):
|
||||
"""
|
||||
Output type of :class:`~transformers.XLNetModel`.
|
||||
Output type of :class:`~transformers.XLNetLMHeadModel`.
|
||||
|
||||
Args:
|
||||
loss (:obj:`torch.FloatTensor` of shape `(1,)`, `optional`, returned when ``labels`` is provided)
|
||||
@@ -637,7 +637,7 @@ class XLNetLMHeadModelOutput(ModelOutput):
|
||||
@dataclass
|
||||
class XLNetForSequenceClassificationOutput(ModelOutput):
|
||||
"""
|
||||
Base class for outputs of sentence classification models.
|
||||
Output type of :class:`~transformers.XLNetForSequenceClassification`.
|
||||
|
||||
Args:
|
||||
loss (:obj:`torch.FloatTensor` of shape :obj:`(1,)`, `optional`, returned when :obj:`label` is provided):
|
||||
@@ -671,7 +671,7 @@ class XLNetForSequenceClassificationOutput(ModelOutput):
|
||||
@dataclass
|
||||
class XLNetForTokenClassificationOutput(ModelOutput):
|
||||
"""
|
||||
Base class for outputs of token classification models.
|
||||
Output type of :class:`~transformers.XLNetForTokenClassificationOutput`.
|
||||
|
||||
Args:
|
||||
loss (:obj:`torch.FloatTensor` of shape :obj:`(1,)`, `optional`, returned when ``labels`` is provided) :
|
||||
|
||||
@@ -46,6 +46,10 @@ if is_tf_available():
|
||||
TFAutoModelForQuestionAnswering,
|
||||
TFAutoModelForTokenClassification,
|
||||
TFAutoModelWithLMHead,
|
||||
TF_MODEL_WITH_LM_HEAD_MAPPING,
|
||||
TF_MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING,
|
||||
TF_MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING,
|
||||
TF_MODEL_FOR_QUESTION_ANSWERING_MAPPING,
|
||||
)
|
||||
|
||||
if is_torch_available():
|
||||
@@ -57,6 +61,11 @@ if is_torch_available():
|
||||
AutoModelForTokenClassification,
|
||||
AutoModelWithLMHead,
|
||||
AutoModelForSeq2SeqLM,
|
||||
MODEL_WITH_LM_HEAD_MAPPING,
|
||||
MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING,
|
||||
MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING,
|
||||
MODEL_FOR_QUESTION_ANSWERING_MAPPING,
|
||||
MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -396,6 +405,7 @@ class Pipeline(_ScikitCompat):
|
||||
if framework is None:
|
||||
framework = get_framework()
|
||||
|
||||
self.task = task
|
||||
self.model = model
|
||||
self.tokenizer = tokenizer
|
||||
self.modelcard = modelcard
|
||||
@@ -469,6 +479,19 @@ class Pipeline(_ScikitCompat):
|
||||
"""
|
||||
return {name: tensor.to(self.device) for name, tensor in inputs.items()}
|
||||
|
||||
def check_model_type(self, supported_models):
|
||||
"""
|
||||
Check if the model class is in the supported class list of the pipeline.
|
||||
"""
|
||||
if not isinstance(supported_models, list): # Create from a model mapping
|
||||
supported_models = [item[1].__name__ for item in supported_models.items()]
|
||||
if self.model.__class__.__name__ not in supported_models:
|
||||
raise PipelineException(
|
||||
self.task,
|
||||
self.model.base_model_prefix,
|
||||
f"The model '{self.model.__class__.__name__}' is not supported for {self.task}. Supported models are {supported_models}",
|
||||
)
|
||||
|
||||
def _parse_and_tokenize(self, *args, padding=True, add_special_tokens=True, **kwargs):
|
||||
"""
|
||||
Parse arguments and tokenize
|
||||
@@ -615,6 +638,11 @@ class TextGenerationPipeline(Pipeline):
|
||||
"TFCTRLLMHeadModel",
|
||||
]
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
self.check_model_type(self.ALLOWED_MODELS)
|
||||
|
||||
# overriding _parse_and_tokenize to allow for unusual language-modeling tokenizer arguments
|
||||
|
||||
def _parse_and_tokenize(self, *args, padding=True, add_special_tokens=True, **kwargs):
|
||||
@@ -640,12 +668,6 @@ class TextGenerationPipeline(Pipeline):
|
||||
def __call__(
|
||||
self, *args, return_tensors=False, return_text=True, clean_up_tokenization_spaces=False, **generate_kwargs
|
||||
):
|
||||
if self.model.__class__.__name__ not in self.ALLOWED_MODELS:
|
||||
raise NotImplementedError(
|
||||
"Generation is currently not supported for {}. Please select a model from {} for generation.".format(
|
||||
self.model.__class__.__name__, self.ALLOWED_MODELS
|
||||
)
|
||||
)
|
||||
|
||||
text_inputs = self._args_parser(*args)
|
||||
|
||||
@@ -771,6 +793,12 @@ class TextClassificationPipeline(Pipeline):
|
||||
def __init__(self, return_all_scores: bool = False, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING
|
||||
if self.framework == "tf"
|
||||
else MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING
|
||||
)
|
||||
|
||||
self.return_all_scores = return_all_scores
|
||||
|
||||
def __call__(self, *args, **kwargs):
|
||||
@@ -847,6 +875,8 @@ class FillMaskPipeline(Pipeline):
|
||||
task=task,
|
||||
)
|
||||
|
||||
self.check_model_type(TF_MODEL_WITH_LM_HEAD_MAPPING if self.framework == "tf" else MODEL_WITH_LM_HEAD_MAPPING)
|
||||
|
||||
self.topk = topk
|
||||
|
||||
def ensure_exactly_one_mask_token(self, masked_index: np.ndarray):
|
||||
@@ -980,6 +1010,12 @@ class TokenClassificationPipeline(Pipeline):
|
||||
task=task,
|
||||
)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING
|
||||
if self.framework == "tf"
|
||||
else MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING
|
||||
)
|
||||
|
||||
self._basic_tokenizer = BasicTokenizer(do_lower_case=False)
|
||||
self.ignore_labels = ignore_labels
|
||||
self.grouped_entities = grouped_entities
|
||||
@@ -1220,6 +1256,10 @@ class QuestionAnsweringPipeline(Pipeline):
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_FOR_QUESTION_ANSWERING_MAPPING if self.framework == "tf" else MODEL_FOR_QUESTION_ANSWERING_MAPPING
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def create_sample(
|
||||
question: Union[str, List[str]], context: Union[str, List[str]]
|
||||
@@ -1483,9 +1523,13 @@ class SummarizationPipeline(Pipeline):
|
||||
on the associated CUDA device id.
|
||||
"""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
def __init__(self, *args, **kwargs):
|
||||
kwargs.update(task="summarization")
|
||||
super().__init__(**kwargs)
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
self.check_model_type(
|
||||
TF_MODEL_WITH_LM_HEAD_MAPPING if self.framework == "tf" else MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING
|
||||
)
|
||||
|
||||
def __call__(
|
||||
self, *documents, return_tensors=False, return_text=True, clean_up_tokenization_spaces=False, **generate_kwargs
|
||||
@@ -1615,6 +1659,11 @@ class TranslationPipeline(Pipeline):
|
||||
on the associated CUDA device id.
|
||||
"""
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
self.check_model_type(TF_MODEL_WITH_LM_HEAD_MAPPING if self.framework == "tf" else MODEL_WITH_LM_HEAD_MAPPING)
|
||||
|
||||
def __call__(
|
||||
self, *args, return_tensors=False, return_text=True, clean_up_tokenization_spaces=False, **generate_kwargs
|
||||
):
|
||||
|
||||
@@ -61,7 +61,7 @@ PRETRAINED_POSITIONAL_EMBEDDINGS_SIZES = {
|
||||
|
||||
class T5Tokenizer(PreTrainedTokenizer):
|
||||
"""
|
||||
Constructs an XLNet tokenizer. Based on `SentencePiece <https://github.com/google/sentencepiece>`__ .
|
||||
Constructs a T5 tokenizer. Based on `SentencePiece <https://github.com/google/sentencepiece>`__ .
|
||||
|
||||
This tokenizer inherits from :class:`~transformers.PreTrainedTokenizer` which contains most of the methods. Users
|
||||
should refer to the superclass for more information regarding methods.
|
||||
|
||||
@@ -618,6 +618,9 @@ class Trainer:
|
||||
|
||||
if self.args.past_index >= 0 and self._past is not None:
|
||||
inputs["mems"] = self._past
|
||||
# Our model outputs do not work with DataParallel, so forcing return tuple.
|
||||
if self.args.n_gpu > 1:
|
||||
inputs["return_tuple"] = True
|
||||
|
||||
outputs = model(**inputs)
|
||||
loss = outputs[0] # model outputs are always tuple in transformers (see doc)
|
||||
@@ -818,6 +821,9 @@ class Trainer:
|
||||
inputs[k] = v.to(self.args.device)
|
||||
if self.args.past_index >= 0:
|
||||
inputs["mems"] = past
|
||||
# Our model outputs do not work with DataParallel, so forcing return tuple.
|
||||
if self.args.n_gpu > 1:
|
||||
inputs["return_tuple"] = True
|
||||
|
||||
with torch.no_grad():
|
||||
outputs = model(**inputs)
|
||||
|
||||
@@ -31,6 +31,7 @@ if is_torch_available():
|
||||
FlaubertForQuestionAnswering,
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
)
|
||||
from transformers.modeling_flaubert import FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST
|
||||
|
||||
@@ -294,6 +295,30 @@ class FlaubertModelTester(object):
|
||||
self.parent.assertListEqual(list(result["loss"].size()), [])
|
||||
self.parent.assertListEqual(list(result["logits"].size()), [self.batch_size, self.type_sequence_label_size])
|
||||
|
||||
def create_and_check_flaubert_token_classif(
|
||||
self,
|
||||
config,
|
||||
input_ids,
|
||||
token_type_ids,
|
||||
input_lengths,
|
||||
sequence_labels,
|
||||
token_labels,
|
||||
is_impossible_labels,
|
||||
input_mask,
|
||||
):
|
||||
config.num_labels = self.num_labels
|
||||
model = FlaubertForTokenClassification(config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
|
||||
loss, logits = model(input_ids, attention_mask=input_mask, labels=token_labels)
|
||||
result = {
|
||||
"loss": loss,
|
||||
"logits": logits,
|
||||
}
|
||||
self.parent.assertListEqual(list(result["logits"].size()), [self.batch_size, self.seq_length, self.num_labels])
|
||||
self.check_loss_output(result)
|
||||
|
||||
def prepare_config_and_inputs_for_common(self):
|
||||
config_and_inputs = self.prepare_config_and_inputs()
|
||||
(
|
||||
@@ -320,6 +345,7 @@ class FlaubertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
FlaubertForQuestionAnswering,
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForTokenClassification,
|
||||
)
|
||||
if is_torch_available()
|
||||
else ()
|
||||
@@ -352,6 +378,10 @@ class FlaubertModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_flaubert_sequence_classif(*config_and_inputs)
|
||||
|
||||
def test_flaubert_token_classif(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_flaubert_token_classif(*config_and_inputs)
|
||||
|
||||
@slow
|
||||
def test_model_from_pretrained(self):
|
||||
for model_name in FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST[:1]:
|
||||
|
||||
@@ -51,7 +51,7 @@ class MBartEnroIntegrationTest(AbstractMBartIntegrationTest):
|
||||
]
|
||||
tgt_text = [
|
||||
"Şeful ONU declară că nu există o soluţie militară în Siria",
|
||||
'Secretarul General Ban Ki-moon declară că răspunsul său la intensificarea sprijinului militar al Rusiei pentru Siria este că "nu există o soluţie militară" la conflictul de aproape cinci ani şi că noi arme nu vor face decât să înrăutăţească violenţele şi mizeria pentru milioane de oameni.',
|
||||
'Secretarul General Ban Ki-moon declară că răspunsul său la intensificarea sprijinului militar al Rusiei pentru Siria este că "nu există o soluţie militară" la conflictul de aproape cinci ani şi că noi arme nu vor face decât să înrăutăţească violenţa şi mizeria pentru milioane de oameni.',
|
||||
]
|
||||
expected_src_tokens = [8274, 127873, 25916, 7, 8622, 2071, 438, 67485, 53, 187895, 23, 51712, 2, EN_CODE]
|
||||
|
||||
|
||||
+125
-15
@@ -28,6 +28,7 @@ if is_torch_available():
|
||||
ReformerForMaskedLM,
|
||||
ReformerModel,
|
||||
ReformerModelWithLMHead,
|
||||
ReformerForSequenceClassification,
|
||||
ReformerTokenizer,
|
||||
ReformerLayer,
|
||||
ReformerForQuestionAnswering,
|
||||
@@ -77,6 +78,7 @@ class ReformerModelTester:
|
||||
eos_token_id=None,
|
||||
scope=None,
|
||||
hash_seed=None,
|
||||
num_labels=None,
|
||||
):
|
||||
self.parent = parent
|
||||
self.batch_size = batch_size
|
||||
@@ -124,6 +126,7 @@ class ReformerModelTester:
|
||||
self.encoder_seq_length = seq_length // attn_chunk_length + (self.seq_length % attn_chunk_length != 0)
|
||||
self.key_length = (num_chunks_before + num_chunks_after + 1) * attn_chunk_length
|
||||
self.chunk_length = attn_chunk_length
|
||||
self.num_labels = num_labels
|
||||
|
||||
def prepare_config_and_inputs(self):
|
||||
input_ids = ids_tensor([self.batch_size, self.seq_length], self.vocab_size)
|
||||
@@ -178,8 +181,8 @@ class ReformerModelTester:
|
||||
model = ReformerModel(config=config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
(sequence_output,) = model(input_ids, attention_mask=input_mask)
|
||||
(sequence_output,) = model(input_ids)
|
||||
sequence_output, _ = model(input_ids, attention_mask=input_mask)
|
||||
sequence_output, _ = model(input_ids)
|
||||
|
||||
result = {
|
||||
"sequence_output": sequence_output,
|
||||
@@ -190,17 +193,21 @@ class ReformerModelTester:
|
||||
)
|
||||
|
||||
def create_and_check_reformer_model_with_lm_backward(self, config, input_ids, input_mask, choice_labels):
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
config.is_decoder = False
|
||||
config.lsh_num_chunks_after = 1
|
||||
model = ReformerForMaskedLM(config=config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
loss = model(input_ids, attention_mask=input_mask, labels=input_ids)[0]
|
||||
loss.backward()
|
||||
|
||||
def create_and_check_reformer_with_lm(self, config, input_ids, input_mask, choice_labels):
|
||||
config.lsh_num_chunks_after = 0
|
||||
config.is_decoder = True
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
loss, prediction_scores = model(input_ids, attention_mask=input_mask, labels=input_ids)
|
||||
loss, prediction_scores, _ = model(input_ids, attention_mask=input_mask, labels=input_ids)
|
||||
result = {
|
||||
"loss": loss,
|
||||
"prediction_scores": prediction_scores,
|
||||
@@ -329,9 +336,11 @@ class ReformerModelTester:
|
||||
config.hidden_dropout_prob = 0
|
||||
config.local_attention_probs_dropout_prob = 0
|
||||
config.lsh_attention_probs_dropout_prob = 0
|
||||
config.lsh_num_chunks_after = 1
|
||||
config.is_decoder = False
|
||||
|
||||
torch.manual_seed(0)
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
model = ReformerForMaskedLM(config=config)
|
||||
model.to(torch_device)
|
||||
model.train()
|
||||
model.zero_grad()
|
||||
@@ -345,7 +354,7 @@ class ReformerModelTester:
|
||||
config.chunk_size_feed_forward = 1
|
||||
|
||||
torch.manual_seed(0)
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
model = ReformerForMaskedLM(config=config)
|
||||
model.to(torch_device)
|
||||
model.train()
|
||||
model.zero_grad()
|
||||
@@ -402,7 +411,22 @@ class ReformerModelTester:
|
||||
output = model(input_ids, attention_mask=input_mask)[0]
|
||||
self.parent.assertFalse(torch.isnan(output).any().item())
|
||||
|
||||
def create_and_check_reformer_model_generate(self, config, input_ids, input_mask, choice_labels):
|
||||
config.is_decoder = True
|
||||
config.lsh_num_chunks_after = 0
|
||||
config.bos_token_id = 0
|
||||
config.eos_token_id = None
|
||||
config.max_length = 20
|
||||
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
output = model.generate()
|
||||
self.parent.assertIsNotNone(output)
|
||||
|
||||
def create_and_check_reformer_model_fp16_generate(self, config, input_ids, input_mask, choice_labels):
|
||||
config.is_decoder = True
|
||||
config.lsh_num_chunks_after = 0
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
model.to(torch_device)
|
||||
model.half()
|
||||
@@ -415,13 +439,15 @@ class ReformerModelTester:
|
||||
# force chunk length to be bigger than input_ids
|
||||
config.lsh_attn_chunk_length = 2 * input_ids.shape[-1]
|
||||
config.local_attn_chunk_length = 2 * input_ids.shape[-1]
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
config.lsh_num_chunks_after = 1
|
||||
config.is_decoder = False
|
||||
model = ReformerForMaskedLM(config=config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
output_logits = model(input_ids, attention_mask=input_mask)[0]
|
||||
self.parent.assertTrue(output_logits.shape[1] == input_ids.shape[-1])
|
||||
|
||||
def create_and_check_longformer_for_question_answering(self, config, input_ids, input_mask, choice_labels):
|
||||
def create_and_check_reformer_for_question_answering(self, config, input_ids, input_mask, choice_labels):
|
||||
model = ReformerForQuestionAnswering(config=config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
@@ -437,12 +463,57 @@ class ReformerModelTester:
|
||||
self.parent.assertListEqual(list(result["end_logits"].size()), [self.batch_size, self.seq_length])
|
||||
self.check_loss_output(result)
|
||||
|
||||
def create_and_check_cached_hidden_states_and_buckets(self, config, input_ids, input_mask, choice_labels):
|
||||
config.is_decoder = True
|
||||
config.lsh_num_chunks_before = 1
|
||||
config.lsh_num_chunks_after = 0
|
||||
model = ReformerModelWithLMHead(config=config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
input_ids_first = input_ids[:, :-1]
|
||||
input_ids_second = input_ids[:, -1:]
|
||||
|
||||
# return saved cache
|
||||
_, cached_hidden_states_and_buckets = model(input_ids_first, use_cache=True)
|
||||
|
||||
# calculate last output with and without cache
|
||||
outputs_with_cache, _ = model(
|
||||
input_ids_second, cached_hidden_states_and_buckets=cached_hidden_states_and_buckets, use_cache=True
|
||||
)
|
||||
outputs_without_cache = model(input_ids)[0][:, -1]
|
||||
|
||||
# select random slice idx
|
||||
random_slice_idx = torch.randint(outputs_without_cache.shape[-1], (1, 1), device=torch_device).item()
|
||||
|
||||
# outputs should be similar within range
|
||||
self.parent.assertTrue(
|
||||
torch.allclose(
|
||||
outputs_with_cache[:, 0, random_slice_idx], outputs_without_cache[:, random_slice_idx], atol=1e-2
|
||||
)
|
||||
)
|
||||
|
||||
def prepare_config_and_inputs_for_common(self):
|
||||
config_and_inputs = self.prepare_config_and_inputs()
|
||||
(config, input_ids, input_mask, choice_labels) = config_and_inputs
|
||||
inputs_dict = {"input_ids": input_ids, "attention_mask": input_mask}
|
||||
return config, inputs_dict
|
||||
|
||||
def create_and_check_reformer_for_sequence_classification(
|
||||
self, config, input_ids, input_mask, choice_labels, is_decoder
|
||||
):
|
||||
config.is_decoder = is_decoder
|
||||
sequence_labels = ids_tensor([self.batch_size], config.num_labels)
|
||||
model = ReformerForSequenceClassification(config)
|
||||
model.to(torch_device)
|
||||
model.eval()
|
||||
loss, logits = model(input_ids, attention_mask=input_mask, labels=sequence_labels)
|
||||
result = {
|
||||
"loss": loss,
|
||||
"logits": logits,
|
||||
}
|
||||
self.parent.assertListEqual(list(result["logits"].size()), [self.batch_size, self.num_labels])
|
||||
self.check_loss_output(result)
|
||||
|
||||
|
||||
class ReformerTesterMixin:
|
||||
"""
|
||||
@@ -490,6 +561,18 @@ class ReformerTesterMixin:
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_reformer_no_chunking(*config_and_inputs)
|
||||
|
||||
def test_reformer_qa_answering(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_reformer_for_question_answering(*config_and_inputs)
|
||||
|
||||
def test_reformer_cached_inference(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_cached_hidden_states_and_buckets(*config_and_inputs)
|
||||
|
||||
def test_reformer_cached_generate(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_reformer_model_generate(*config_and_inputs)
|
||||
|
||||
@slow
|
||||
def test_dropout_random_seed_is_changing(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
@@ -510,11 +593,17 @@ class ReformerTesterMixin:
|
||||
# Opt-out of this test.
|
||||
pass
|
||||
|
||||
def test_for_sequence_classification(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_reformer_for_sequence_classification(*config_and_inputs, is_decoder=False)
|
||||
|
||||
|
||||
@require_torch
|
||||
class ReformerLocalAttnModelTest(ReformerTesterMixin, ModelTesterMixin, unittest.TestCase):
|
||||
all_model_classes = (
|
||||
(ReformerModel, ReformerModelWithLMHead, ReformerForQuestionAnswering) if is_torch_available() else ()
|
||||
(ReformerModel, ReformerModelWithLMHead, ReformerForSequenceClassification, ReformerForQuestionAnswering)
|
||||
if is_torch_available()
|
||||
else ()
|
||||
)
|
||||
all_generative_model_classes = (ReformerModelWithLMHead,) if is_torch_available() else ()
|
||||
test_pruning = False
|
||||
@@ -554,6 +643,7 @@ class ReformerLocalAttnModelTest(ReformerTesterMixin, ModelTesterMixin, unittest
|
||||
"eos_token_id": 2,
|
||||
"scope": None,
|
||||
"hash_seed": 0,
|
||||
"num_labels": 2,
|
||||
}
|
||||
|
||||
def setUp(self):
|
||||
@@ -571,7 +661,9 @@ class ReformerLocalAttnModelTest(ReformerTesterMixin, ModelTesterMixin, unittest
|
||||
@require_torch
|
||||
class ReformerLSHAttnModelTest(ReformerTesterMixin, ModelTesterMixin, unittest.TestCase):
|
||||
all_model_classes = (
|
||||
(ReformerModel, ReformerModelWithLMHead, ReformerForQuestionAnswering) if is_torch_available() else ()
|
||||
(ReformerModel, ReformerModelWithLMHead, ReformerForSequenceClassification, ReformerForQuestionAnswering)
|
||||
if is_torch_available()
|
||||
else ()
|
||||
)
|
||||
all_generative_model_classes = (ReformerModelWithLMHead,) if is_torch_available() else ()
|
||||
test_pruning = False
|
||||
@@ -593,8 +685,8 @@ class ReformerLSHAttnModelTest(ReformerTesterMixin, ModelTesterMixin, unittest.T
|
||||
"num_buckets": 2,
|
||||
"num_hashes": 4,
|
||||
"lsh_attn_chunk_length": 4,
|
||||
"lsh_num_chunks_before": 2,
|
||||
"lsh_num_chunks_after": 3,
|
||||
"lsh_num_chunks_before": 1,
|
||||
"lsh_num_chunks_after": 0,
|
||||
"chunk_size_lm_head": 5,
|
||||
"chunk_size_feed_forward": 6,
|
||||
"feed_forward_size": 32,
|
||||
@@ -608,11 +700,14 @@ class ReformerLSHAttnModelTest(ReformerTesterMixin, ModelTesterMixin, unittest.T
|
||||
"axial_pos_embds": True,
|
||||
"axial_pos_shape": [4, 8],
|
||||
"axial_pos_embds_dim": [16, 48],
|
||||
"attn_layers": ["lsh", "lsh", "lsh", "lsh"],
|
||||
# sanotheu
|
||||
# "attn_layers": ["lsh", "lsh", "lsh", "lsh"],
|
||||
"attn_layers": ["lsh"],
|
||||
"pad_token_id": 0,
|
||||
"eos_token_id": 2,
|
||||
"scope": None,
|
||||
"hash_seed": 0,
|
||||
"num_labels": 2,
|
||||
}
|
||||
|
||||
def setUp(self):
|
||||
@@ -1020,8 +1115,23 @@ class ReformerIntegrationTests(unittest.TestCase):
|
||||
output_ids = model.generate(
|
||||
input_ids, max_length=50, num_beams=4, early_stopping=True, do_sample=False, num_hashes=8
|
||||
)
|
||||
output_text = tokenizer.decode(output_ids[0])
|
||||
output = tokenizer.decode(output_ids[0])
|
||||
|
||||
self.assertEqual(
|
||||
output_text,
|
||||
output,
|
||||
"A few months later state expression in his ideas, at the first entrance. He was positively for an inst",
|
||||
)
|
||||
|
||||
@slow
|
||||
def test_pretrained_generate_use_cache_equality(self):
|
||||
model = ReformerModelWithLMHead.from_pretrained("google/reformer-crime-and-punishment").to(torch_device)
|
||||
tokenizer = ReformerTokenizer.from_pretrained("google/reformer-crime-and-punishment")
|
||||
model.eval()
|
||||
input_ids = tokenizer.encode("A few months later", return_tensors="pt").to(torch_device)
|
||||
output_ids_with_cache = model.generate(input_ids, max_length=130, num_hashes=8, use_cache=False)
|
||||
output_ids_without_cache = model.generate(input_ids, max_length=130, num_hashes=8, use_cache=True)
|
||||
|
||||
output_with_cache = tokenizer.decode(output_ids_with_cache[0])
|
||||
output_without_cache = tokenizer.decode(output_ids_without_cache[0])
|
||||
|
||||
self.assertEqual(output_with_cache, output_without_cache)
|
||||
|
||||
@@ -297,7 +297,7 @@ class XLMModelTester:
|
||||
self.parent.assertListEqual(list(result["loss"].size()), [])
|
||||
self.parent.assertListEqual(list(result["logits"].size()), [self.batch_size, self.type_sequence_label_size])
|
||||
|
||||
def create_and_check_xlm_for_token_classification(
|
||||
def create_and_check_xlm_token_classif(
|
||||
self,
|
||||
config,
|
||||
input_ids,
|
||||
@@ -383,9 +383,9 @@ class XLMModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_xlm_sequence_classif(*config_and_inputs)
|
||||
|
||||
def test_xlm_for_token_classification(self):
|
||||
def test_xlm_token_classif(self):
|
||||
config_and_inputs = self.model_tester.prepare_config_and_inputs()
|
||||
self.model_tester.create_and_check_xlm_for_token_classification(*config_and_inputs)
|
||||
self.model_tester.create_and_check_xlm_token_classif(*config_and_inputs)
|
||||
|
||||
@slow
|
||||
def test_model_from_pretrained(self):
|
||||
|
||||
Reference in New Issue
Block a user