Compare commits
68
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b9b777749b | ||
|
|
79eb391586 | ||
|
|
7087d9b1c0 | ||
|
|
efc4a21ffa | ||
|
|
5148f43309 | ||
|
|
38f6739cd6 | ||
|
|
00602f7840 | ||
|
|
3c682ea15c | ||
|
|
59b5953d89 | ||
|
|
6e07c1f446 | ||
|
|
43fdafef89 | ||
|
|
627e813734 | ||
|
|
9865e1fe52 | ||
|
|
d39da5a2ab | ||
|
|
5e323017a4 | ||
|
|
4acfd1a8dc | ||
|
|
3a40cdf58d | ||
|
|
88b3a91e61 | ||
|
|
023f0f3708 | ||
|
|
64b24bb3c2 | ||
|
|
0397619ac6 | ||
|
|
5ac07513e0 | ||
|
|
5ae935d233 | ||
|
|
467573ddde | ||
|
|
077c99bb5f | ||
|
|
06fc3954a1 | ||
|
|
ff65beafa3 | ||
|
|
2e5052d4f1 | ||
|
|
18ce6b8ff3 | ||
|
|
901e9b8eda | ||
|
|
f34372a9ff | ||
|
|
cc2e312ca3 | ||
|
|
a16e568f22 | ||
|
|
64b4d25cf3 | ||
|
|
3479787edc | ||
|
|
a7db81c33f | ||
|
|
f774b2e8c4 | ||
|
|
8348105692 | ||
|
|
95792a948e | ||
|
|
4abb7ffc18 | ||
|
|
8b38173398 | ||
|
|
f8d3695e8c | ||
|
|
16da877139 | ||
|
|
52decab371 | ||
|
|
9b6610f7f6 | ||
|
|
e174bfeb34 | ||
|
|
bf162ce8ca | ||
|
|
58fb25f25b | ||
|
|
2b07ec7823 | ||
|
|
35d2ad5b83 | ||
|
|
bdda4f2249 | ||
|
|
8e23749649 | ||
|
|
3eaa007d78 | ||
|
|
758572cad8 | ||
|
|
57516c0cc8 | ||
|
|
006a16483f | ||
|
|
16d3cc187d | ||
|
|
829842159e | ||
|
|
5cd9e2cba1 | ||
|
|
220b5f97ca | ||
|
|
8ffd7fb12d | ||
|
|
613ab364eb | ||
|
|
f7eb17dc47 | ||
|
|
29792864cb | ||
|
|
13842e413c | ||
|
|
0e24e4c136 | ||
|
|
96f4828ace | ||
|
|
ef0ac063c9 |
+53
-1
@@ -84,7 +84,7 @@ jobs:
|
||||
key: v0.3-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -rA -s ./tests/ --cov --durations=0 | tee output.txt
|
||||
- run: RUN_PT_TF_CROSS_TESTS=1 python -m pytest -n 8 --dist=loadfile -rA -s ./tests/ -m is_pt_tf_cross_test --cov --durations=0 | tee output.txt
|
||||
- run: codecov
|
||||
- store_artifacts:
|
||||
path: ~/transformers/output.txt
|
||||
@@ -164,6 +164,56 @@ jobs:
|
||||
- store_artifacts:
|
||||
path: ~/transformers/output.txt
|
||||
destination: test_output.txt
|
||||
run_tests_pipelines_torch:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.3-torch-{{ checksum "setup.py" }}
|
||||
- v0.3-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install git+https://github.com/huggingface/datasets
|
||||
- run: pip install .[sklearn,torch,testing]
|
||||
- save_cache:
|
||||
key: v0.3-torch-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: RUN_PIPELINE_TESTS=1 python -m pytest -n 8 --dist=loadfile -rA -s ./tests/ -m is_pipeline_test | tee output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/output.txt
|
||||
destination: test_output.txt
|
||||
run_tests_pipelines_tf:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.3-tf-{{ checksum "setup.py" }}
|
||||
- v0.3-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install git+https://github.com/huggingface/datasets
|
||||
- run: pip install .[sklearn,tf-cpu,testing]
|
||||
- save_cache:
|
||||
key: v0.3-tf-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: RUN_PIPELINE_TESTS=1 python -m pytest -n 8 --dist=loadfile -rA -s ./tests/ -m is_pipeline_test | tee output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/output.txt
|
||||
destination: test_output.txt
|
||||
run_tests_custom_tokenizers:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
@@ -331,6 +381,8 @@ workflows:
|
||||
- run_tests_torch
|
||||
- run_tests_tf
|
||||
- run_tests_flax
|
||||
- run_tests_pipelines_torch
|
||||
- run_tests_pipelines_tf
|
||||
- build_doc
|
||||
- deploy_doc: *workflow_filters
|
||||
tpu_testing_jobs:
|
||||
|
||||
+2
-1
@@ -50,4 +50,5 @@ deploy_doc "b42586e" v2.11.0
|
||||
deploy_doc "7fb8bdf" v3.0.2
|
||||
deploy_doc "4b3ee9c" v3.1.0
|
||||
deploy_doc "3ebb1b3" v3.2.0
|
||||
deploy_doc "0613f05" # v3.3.0 Latest stable release
|
||||
deploy_doc "0613f05" v3.3.1
|
||||
deploy_doc "eb0e0ce" # v3.4.0 Latest stable release
|
||||
|
||||
@@ -16,52 +16,52 @@ jobs:
|
||||
run_tests_torch_and_tf_gpu:
|
||||
runs-on: [self-hosted, single-gpu]
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- name: Python version
|
||||
run: |
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Current dir
|
||||
run: pwd
|
||||
- run: nvidia-smi
|
||||
- uses: actions/checkout@v2
|
||||
- name: Python version
|
||||
run: |
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Current dir
|
||||
run: pwd
|
||||
- run: nvidia-smi
|
||||
|
||||
- name: Loading cache.
|
||||
uses: actions/cache@v2
|
||||
id: cache
|
||||
with:
|
||||
path: .env
|
||||
key: v0-tests_tf_torch_gpu-${{ hashFiles('setup.py') }}
|
||||
- name: Loading cache.
|
||||
uses: actions/cache@v2
|
||||
id: cache
|
||||
with:
|
||||
path: .env
|
||||
key: v0-tests_tf_torch_gpu-${{ hashFiles('setup.py') }}
|
||||
|
||||
- name: Create new python env (on self-hosted runners we have to handle isolation ourselves)
|
||||
run: |
|
||||
python -m venv .env
|
||||
source .env/bin/activate
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
pip install --upgrade pip
|
||||
pip install torch!=1.6.0
|
||||
pip install .[sklearn,testing,onnxruntime]
|
||||
pip install git+https://github.com/huggingface/datasets
|
||||
- name: Create new python env (on self-hosted runners we have to handle isolation ourselves)
|
||||
run: |
|
||||
python -m venv .env
|
||||
source .env/bin/activate
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
pip install --upgrade pip
|
||||
pip install torch!=1.6.0
|
||||
pip install .[sklearn,testing,onnxruntime]
|
||||
pip install git+https://github.com/huggingface/datasets
|
||||
|
||||
- name: Are GPUs recognized by our DL frameworks
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -c "import torch; print('Cuda available:', torch.cuda.is_available())"
|
||||
python -c "import torch; print('Number of GPUs available:', torch.cuda.device_count())"
|
||||
- name: Are GPUs recognized by our DL frameworks
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -c "import torch; print('Cuda available:', torch.cuda.is_available())"
|
||||
python -c "import torch; print('Number of GPUs available:', torch.cuda.device_count())"
|
||||
|
||||
- name: Run all non-slow tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
# TF_GPU_MEMORY_LIMIT: 4096
|
||||
OMP_NUM_THREADS: 1
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 2 --dist=loadfile -s ./tests/
|
||||
- name: Run all non-slow tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
# TF_GPU_MEMORY_LIMIT: 4096
|
||||
OMP_NUM_THREADS: 1
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 2 --dist=loadfile -s ./tests/
|
||||
|
||||
run_tests_torch_and_tf_multiple_gpu:
|
||||
runs-on: [self-hosted, multi-gpu]
|
||||
|
||||
@@ -12,64 +12,75 @@ jobs:
|
||||
run_all_tests_torch_and_tf_gpu:
|
||||
runs-on: [self-hosted, single-gpu]
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- name: Loading cache.
|
||||
uses: actions/cache@v2
|
||||
id: cache
|
||||
with:
|
||||
path: .env
|
||||
key: v0-slow_tests_tf_torch_gpu-${{ hashFiles('setup.py') }}
|
||||
- name: Loading cache.
|
||||
uses: actions/cache@v2
|
||||
id: cache
|
||||
with:
|
||||
path: .env
|
||||
key: v0-slow_tests_tf_torch_gpu-${{ hashFiles('setup.py') }}
|
||||
|
||||
- name: Python version
|
||||
run: |
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Current dir
|
||||
run: pwd
|
||||
- run: nvidia-smi
|
||||
- name: Create new python env (on self-hosted runners we have to handle isolation ourselves)
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
run: |
|
||||
python -m venv .env
|
||||
source .env/bin/activate
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
pip install --upgrade pip
|
||||
pip install torch!=1.6.0
|
||||
pip install .[sklearn,testing,onnxruntime]
|
||||
pip install git+https://github.com/huggingface/datasets
|
||||
- name: Python version
|
||||
run: |
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Current dir
|
||||
run: pwd
|
||||
- run: nvidia-smi
|
||||
- name: Create new python env (on self-hosted runners we have to handle isolation ourselves)
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
run: |
|
||||
python -m venv .env
|
||||
source .env/bin/activate
|
||||
which python
|
||||
python --version
|
||||
pip --version
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
pip install --upgrade pip
|
||||
pip install torch!=1.6.0
|
||||
pip install .[sklearn,testing,onnxruntime]
|
||||
pip install git+https://github.com/huggingface/datasets
|
||||
|
||||
- name: Are GPUs recognized by our DL frameworks
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -c "import torch; print('Cuda available:', torch.cuda.is_available())"
|
||||
python -c "import torch; print('Number of GPUs available:', torch.cuda.device_count())"
|
||||
- name: Are GPUs recognized by our DL frameworks
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -c "import torch; print('Cuda available:', torch.cuda.is_available())"
|
||||
python -c "import torch; print('Number of GPUs available:', torch.cuda.device_count())"
|
||||
|
||||
|
||||
- name: Run all tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
OMP_NUM_THREADS: 1
|
||||
RUN_SLOW: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ --durations=0
|
||||
- name: Run all tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
OMP_NUM_THREADS: 1
|
||||
RUN_SLOW: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ --durations=50
|
||||
|
||||
- name: Run examples tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
OMP_NUM_THREADS: 1
|
||||
RUN_SLOW: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
pip install -r examples/requirements.txt
|
||||
python -m pytest -n 1 --dist=loadfile -s examples --durations=50
|
||||
|
||||
- name: Run all pipeline tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
OMP_NUM_THREADS: 1
|
||||
RUN_SLOW: yes
|
||||
RUN_PIPELINE_TESTS: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ -m is_pipeline_test --durations=50
|
||||
|
||||
- name: Run examples tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
OMP_NUM_THREADS: 1
|
||||
RUN_SLOW: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
pip install -r examples/requirements.txt
|
||||
python -m pytest -n 1 --dist=loadfile -s examples --durations=0
|
||||
|
||||
run_all_tests_torch_and_tf_multiple_gpu:
|
||||
runs-on: [self-hosted, multi-gpu]
|
||||
@@ -120,7 +131,7 @@ jobs:
|
||||
RUN_SLOW: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ --durations=0
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ --durations=50
|
||||
|
||||
- name: Run examples tests on GPU
|
||||
env:
|
||||
@@ -130,4 +141,14 @@ jobs:
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
pip install -r examples/requirements.txt
|
||||
python -m pytest -n 1 --dist=loadfile -s examples --durations=0
|
||||
python -m pytest -n 1 --dist=loadfile -s examples --durations=50
|
||||
|
||||
- name: Run all pipeline tests on GPU
|
||||
env:
|
||||
TF_FORCE_GPU_ALLOW_GROWTH: "true"
|
||||
OMP_NUM_THREADS: 1
|
||||
RUN_SLOW: yes
|
||||
RUN_PIPELINE_TESTS: yes
|
||||
run: |
|
||||
source .env/bin/activate
|
||||
python -m pytest -n 1 --dist=loadfile -s ./tests/ -m is_pipeline_test --durations=50
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
// These two things need to be updated at each release for the version selector.
|
||||
// Last stable version
|
||||
const stableVersion = "v3.3.0"
|
||||
const stableVersion = "v3.4.0"
|
||||
// Dictionary doc folder to label
|
||||
const versionMapping = {
|
||||
"master": "master",
|
||||
"": "v3.3.0/v3.3.1",
|
||||
"": "v3.4.0",
|
||||
"v3.3.1": "v3.3.0/v3.3.1",
|
||||
"v3.2.0": "v3.2.0",
|
||||
"v3.1.0": "v3.1.0 (stable)",
|
||||
"v3.0.2": "v3.0.0/v3.0.1/v3.0.2",
|
||||
|
||||
@@ -86,3 +86,18 @@ BartForQuestionAnswering
|
||||
|
||||
.. autoclass:: transformers.BartForQuestionAnswering
|
||||
:members: forward
|
||||
|
||||
|
||||
|
||||
TFBartModel
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TFBartModel
|
||||
:members: call
|
||||
|
||||
|
||||
TFBartForConditionalGeneration
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TFBartForConditionalGeneration
|
||||
:members: call
|
||||
|
||||
@@ -500,8 +500,8 @@ BART
|
||||
<https://arxiv.org/abs/1910.13461>`_, Mike Lewis et al.
|
||||
|
||||
Sequence-to-sequence model with an encoder and a decoder. Encoder is fed a corrupted version of the tokens, decoder is
|
||||
fed the original tokens (but has a mask to hide the future words like a regular transformers decoder). For the encoder, on the
|
||||
pretraining tasks, a composition of the following transformations are applied:
|
||||
fed the original tokens (but has a mask to hide the future words like a regular transformers decoder). For the encoder
|
||||
, on the pretraining tasks, a composition of the following transformations are applied:
|
||||
|
||||
* mask random tokens (like in BERT)
|
||||
* delete random tokens
|
||||
@@ -526,12 +526,17 @@ Pegasus
|
||||
`PEGASUS: Pre-training with Extracted Gap-sentences forAbstractive Summarization
|
||||
<https://arxiv.org/pdf/1912.08777.pdf>`_, Jingqing Zhang, Yao Zhao, Mohammad Saleh and Peter J. Liu on Dec 18, 2019.
|
||||
|
||||
Sequence-to-sequence model with the same encoder-decoder model architecture as BART. Pegasus is pre-trained jointly on two self-supervised objective functions: Masked Language Modeling (MLM) and a novel summarization specific pre-training objective, called Gap Sentence Generation (GSG).
|
||||
Sequence-to-sequence model with the same encoder-decoder model architecture as BART. Pegasus is pre-trained jointly on
|
||||
two self-supervised objective functions: Masked Language Modeling (MLM) and a novel summarization specific pre-training
|
||||
objective, called Gap Sentence Generation (GSG).
|
||||
|
||||
* MLM: encoder input tokens are randomely replaced by a mask tokens and have to be predicted by the encoder (like in BERT)
|
||||
* GSG: whole encoder input sentences are replaced by a second mask token and fed to the decoder, but which has a causal mask to hide the future words like a regular auto-regressive transformer decoder.
|
||||
* MLM: encoder input tokens are randomely replaced by a mask tokens and have to be predicted by the encoder (like
|
||||
in BERT)
|
||||
* GSG: whole encoder input sentences are replaced by a second mask token and fed to the decoder, but which has a
|
||||
causal mask to hide the future words like a regular auto-regressive transformer decoder.
|
||||
|
||||
In contrast to BART, Pegasus' pretraining task is intentionally similar to summarization: important sentences are masked and are generated together as one output sequence from the remaining sentences, similar to an extractive summary.
|
||||
In contrast to BART, Pegasus' pretraining task is intentionally similar to summarization: important sentences are
|
||||
masked and are generated together as one output sequence from the remaining sentences, similar to an extractive summary.
|
||||
|
||||
The library provides a version of this model for conditional generation, which should be used for summarization.
|
||||
|
||||
@@ -577,11 +582,12 @@ The pretraining includes both supervised and self-supervised training. Supervise
|
||||
tasks provided by the GLUE and SuperGLUE benchmarks (converting them into text-to-text tasks as explained above).
|
||||
|
||||
Self-supervised training uses corrupted tokens, by randomly removing 15% of the tokens and
|
||||
replacing them with individual sentinel tokens (if several consecutive tokens are marked for removal, the whole group is replaced with a single sentinel token). The input of the encoder is the corrupted sentence, the input of the decoder is the
|
||||
original sentence and the target is then the dropped out tokens delimited by their sentinel tokens.
|
||||
replacing them with individual sentinel tokens (if several consecutive tokens are marked for removal, the whole group
|
||||
is replaced with a single sentinel token). The input of the encoder is the corrupted sentence, the input of the decoder
|
||||
is the original sentence and the target is then the dropped out tokens delimited by their sentinel tokens.
|
||||
|
||||
For instance, if we have the sentence “My dog is very cute .”, and we decide to remove the tokens: "dog", "is" and "cute", the encoder
|
||||
input becomes “My <x> very <y> .” and the target input becomes “<x> dog is <y> cute .<z>”
|
||||
For instance, if we have the sentence “My dog is very cute .”, and we decide to remove the tokens: "dog", "is" and
|
||||
"cute", the encoder input becomes “My <x> very <y> .” and the target input becomes “<x> dog is <y> cute .<z>”
|
||||
|
||||
The library provides a version of this model for conditional generation.
|
||||
|
||||
@@ -597,7 +603,8 @@ MBart
|
||||
<img alt="Doc" src="https://img.shields.io/badge/Model_documentation-mbart-blueviolet">
|
||||
</a>
|
||||
|
||||
`Multilingual Denoising Pre-training for Neural Machine Translation <https://arxiv.org/abs/2001.08210>`_ by Yinhan Liu, Jiatao Gu, Naman Goyal, Xian Li, Sergey Edunov
|
||||
`Multilingual Denoising Pre-training for Neural Machine Translation <https://arxiv.org/abs/2001.08210>`_ by Yinhan
|
||||
Liu, Jiatao Gu, Naman Goyal, Xian Li, Sergey Edunov
|
||||
Marjan Ghazvininejad, Mike Lewis, Luke Zettlemoyer.
|
||||
|
||||
The model architecture and pre-training objective is same as BART, but MBart is trained on 25 languages
|
||||
@@ -606,11 +613,12 @@ for pre-training a complete sequence-to-sequence model by denoising full texts i
|
||||
|
||||
The library provides a version of this model for conditional generation.
|
||||
|
||||
The `mbart-large-en-ro checkpoint <https://huggingface.co/facebook/mbart-large-en-ro>`_ can be used for english -> romanian translation.
|
||||
The `mbart-large-en-ro checkpoint <https://huggingface.co/facebook/mbart-large-en-ro>`_ can be used for english ->
|
||||
romanian translation.
|
||||
|
||||
The `mbart-large-cc25 <https://huggingface.co/facebook/mbart-large-cc25>`_ checkpoint can be finetuned for other translation and summarization tasks, using code in ```examples/seq2seq/``` , but is not very useful without finetuning.
|
||||
The `mbart-large-cc25 <https://huggingface.co/facebook/mbart-large-cc25>`_ checkpoint can be finetuned for other
|
||||
translation and summarization tasks, using code in ```examples/seq2seq/``` , but is not very useful without finetuning.
|
||||
|
||||
.. _multimodal-models:
|
||||
|
||||
ProphetNet
|
||||
-----------------------------------------------------------------------------------------------------------------------
|
||||
@@ -624,12 +632,18 @@ ProphetNet
|
||||
<img alt="Doc" src="https://img.shields.io/badge/Model_documentation-prophetnet-blueviolet">
|
||||
</a>
|
||||
|
||||
`ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training, <https://arxiv.org/abs/2001.04063>`__ by Yu Yan, Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang, Ming Zhou.
|
||||
`ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training, <https://arxiv.org/abs/2001.04063>`__ by
|
||||
Yu Yan, Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang, Ming Zhou.
|
||||
|
||||
ProphetNet introduces a novel *sequence-to-sequence* pre-training objective, called *future n-gram prediction*. In future n-gram prediction, the model predicts the next n tokens simultaneously based on previous context tokens at each time step instead instead of just the single next token. The future n-gram prediction explicitly encourages the model to plan for the future tokens and prevent overfitting on strong local correlations.
|
||||
The model architecture is based on the original Transformer, but replaces the "standard" self-attention mechanism in the decoder by a a main self-attention mechanism and a self and n-stream (predict) self-attention mechanism.
|
||||
ProphetNet introduces a novel *sequence-to-sequence* pre-training objective, called *future n-gram prediction*. In
|
||||
future n-gram prediction, the model predicts the next n tokens simultaneously based on previous context tokens at
|
||||
each time step instead instead of just the single next token. The future n-gram prediction explicitly encourages
|
||||
the model to plan for the future tokens and prevent overfitting on strong local correlations.
|
||||
The model architecture is based on the original Transformer, but replaces the "standard" self-attention mechanism
|
||||
in the decoder by a a main self-attention mechanism and a self and n-stream (predict) self-attention mechanism.
|
||||
|
||||
The library provides a pre-trained version of this model for conditional generation and a fine-tuned version for summarization.
|
||||
The library provides a pre-trained version of this model for conditional generation and a fine-tuned version for
|
||||
summarization.
|
||||
|
||||
XLM-ProphetNet
|
||||
-----------------------------------------------------------------------------------------------------------------------
|
||||
@@ -643,11 +657,16 @@ XLM-ProphetNet
|
||||
<img alt="Doc" src="https://img.shields.io/badge/Model_documentation-xprophetnet-blueviolet">
|
||||
</a>
|
||||
|
||||
`ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training, <https://arxiv.org/abs/2001.04063>`__ by Yu Yan, Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang, Ming Zhou.
|
||||
`ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training, <https://arxiv.org/abs/2001.04063>`__ by
|
||||
Yu Yan, Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang, Ming Zhou.
|
||||
|
||||
XLM-ProphetNet's model architecture and pre-training objective is same as ProphetNet, but XLM-ProphetNet was pre-trained on the cross-lingual dataset `XGLUE <https://arxiv.org/abs/2004.01401>`__.
|
||||
XLM-ProphetNet's model architecture and pre-training objective is same as ProphetNet, but XLM-ProphetNet was
|
||||
pre-trained on the cross-lingual dataset `XGLUE <https://arxiv.org/abs/2004.01401>`__.
|
||||
|
||||
The library provides a pre-trained version of this model for multi-lingual conditional generation and fine-tuned versions for headline generation and question generation, respectively.
|
||||
The library provides a pre-trained version of this model for multi-lingual conditional generation and fine-tuned
|
||||
versions for headline generation and question generation, respectively.
|
||||
|
||||
.. _multimodal-models:
|
||||
|
||||
Multimodal models
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
@@ -125,18 +125,19 @@ are 512 preceding tokens available to condition on).
|
||||
lls = []
|
||||
for i in tqdm(range(0, encodings.input_ids.size(1), stride)):
|
||||
begin_loc = max(i + stride - max_length, 0)
|
||||
end_loc = i + stride
|
||||
end_loc = min(i + stride, encodings.input_ids.size(1))
|
||||
trg_len = end_loc - i # may be different from stride on last loop
|
||||
input_ids = encodings.input_ids[:,begin_loc:end_loc].to(device)
|
||||
target_ids = input_ids.clone()
|
||||
target_ids[:,:-stride] = -100
|
||||
target_ids[:,:-trg_len] = -100
|
||||
|
||||
with torch.no_grad():
|
||||
outputs = model(input_ids, labels=target_ids)
|
||||
log_likelihood = outputs[0] * stride
|
||||
log_likelihood = outputs[0] * trg_len
|
||||
|
||||
lls.append(log_likelihood)
|
||||
|
||||
ppl = torch.exp(torch.stack(lls).sum() / i)
|
||||
|
||||
ppl = torch.exp(torch.stack(lls).sum() / end_loc)
|
||||
|
||||
Running this with the stride length equal to the max input length is
|
||||
equivalent to the suboptimal, non-sliding-window strategy we discussed above.
|
||||
|
||||
+30
-6
@@ -765,12 +765,10 @@ or skip the whole module:
|
||||
|
||||
More details, example and ways are `here <https://docs.pytest.org/en/latest/skipping.html>`__.
|
||||
|
||||
Custom markers
|
||||
Slow tests
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
* Slow tests
|
||||
|
||||
Tests that are too slow (e.g. once downloading huge model files) are marked with:
|
||||
The library of tests is ever-growing, and some of the tests take minutes to run, therefore we can't afford waiting for an hour for the test suite to complete on CI. Therefore, with some exceptions for essential tests, slow tests should be marked as in the example below:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
@@ -778,13 +776,13 @@ Tests that are too slow (e.g. once downloading huge model files) are marked with
|
||||
@slow
|
||||
def test_integration_foo():
|
||||
|
||||
To run such tests set ``RUN_SLOW=1`` env var, e.g.:
|
||||
Once a test is marked as ``@slow``, to run such tests set ``RUN_SLOW=1`` env var, e.g.:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
RUN_SLOW=1 pytest tests
|
||||
|
||||
Some decorators like ``@parametrized`` rewrite test names, therefore ``@slow`` and the rest of the skip decorators ``@require_*`` have to be listed last for them to work correctly. Here is an example of the correct usage:
|
||||
Some decorators like ``@parameterized`` rewrite test names, therefore ``@slow`` and the rest of the skip decorators ``@require_*`` have to be listed last for them to work correctly. Here is an example of the correct usage:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
@@ -792,6 +790,32 @@ Some decorators like ``@parametrized`` rewrite test names, therefore ``@slow`` a
|
||||
@slow
|
||||
def test_integration_foo():
|
||||
|
||||
As explained at the beginning of this document, slow tests get to run on a scheduled basis, rather than in PRs CI checks. So it's possible that some problems will be missed during a PR submission and get merged. Such problems will get caught during the next scheduled CI job. But it also means that it's important to run the slow tests on your machine before submitting the PR.
|
||||
|
||||
Here is a rough decision making mechanism for choosing which tests should be marked as slow:
|
||||
|
||||
If the test is focused on one of the library's internal components (e.g., modeling files, tokenization files, pipelines), then we should run that test in the non-slow test suite. If it's focused on an other aspect of the library, such as the documentation or the examples, then we should run these tests in the slow test suite. And then, to refine this approach we should have exceptions:
|
||||
|
||||
* All tests that need to download a heavy set of weights (e.g., model or tokenizer integration tests, pipeline integration tests) should be set to slow. If you're adding a new model, you should create and upload to the hub a tiny version of it (with random weights) for integration tests. This is discussed in the following paragraphs.
|
||||
* All tests that need to do a training not specifically optimized to be fast should be set to slow.
|
||||
* We can introduce exceptions if some of these should-be-non-slow tests are excruciatingly slow, and set them to ``@slow``. Auto-modeling tests, which save and load large files to disk, are a good example of tests that are marked as ``@slow``.
|
||||
* If a test completes under 1 second on CI (including downloads if any) then it should be a normal test regardless.
|
||||
|
||||
Collectively, all the non-slow tests need to cover entirely the different internals, while remaining fast.
|
||||
For example, a significant coverage can be achieved by testing with specially created tiny models with random weights. Such models have the very minimal number of layers (e.g., 2), vocab size (e.g., 1000), etc.
|
||||
Then the ``@slow`` tests can use large slow models to do qualitative testing. To see the use of these simply look for *tiny* models with:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
grep tiny tests examples
|
||||
|
||||
Here is a an example of a `script <https://github.com/huggingface/transformers/blob/master/scripts/fsmt/fsmt-make-tiny-model.py>`__ that created the tiny model `stas/tiny-wmt19-en-de <https://huggingface.co/stas/tiny-wmt19-en-de>`__. You can easily adjust it to your specific model's architecture.
|
||||
|
||||
It's easy to measure the run-time incorrectly if for example there is an overheard of downloading a huge model, but if you test it locally the downloaded files would be cached and thus the download time not measured. Hence check the execution speed report in CI logs instead (the output of ``pytest --durations=0 tests``).
|
||||
|
||||
That report is also useful to find slow outliers that aren't marked as such, or which need to be re-written to be fast. If you notice that the test suite starts getting slow on CI, the top listing of this report will show the slowest tests.
|
||||
|
||||
|
||||
Testing the stdout/stderr output
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
|
||||
@@ -45,6 +45,8 @@ slightly slower (over-fitting takes more epochs).
|
||||
|
||||
We use the `--mlm` flag so that the script may change its loss function.
|
||||
|
||||
If using whole-word masking, use both the`--mlm` and `--wwm` flags.
|
||||
|
||||
```bash
|
||||
export TRAIN_FILE=/path/to/dataset/wiki.train.raw
|
||||
export TEST_FILE=/path/to/dataset/wiki.test.raw
|
||||
@@ -57,7 +59,55 @@ python run_language_modeling.py \
|
||||
--train_data_file=$TRAIN_FILE \
|
||||
--do_eval \
|
||||
--eval_data_file=$TEST_FILE \
|
||||
--mlm
|
||||
--mlm \
|
||||
--wwm
|
||||
```
|
||||
|
||||
For Chinese models, it's same with English model with only --mlm`. If using whole-word masking, we need to generate a reference files, case it's char level.
|
||||
|
||||
**Q :** Why ref file ?
|
||||
|
||||
**A :** Suppose we have a Chinese sentence like : `我喜欢你` The original Chinese-BERT will tokenize it as `['我','喜','欢','你']` in char level.
|
||||
Actually, `喜欢` is a whole word. For whole word mask proxy, We need res like `['我','喜','##欢','你']`.
|
||||
So we need a ref file to tell model which pos of BERT original token should be added `##`.
|
||||
|
||||
**Q :** Why LTP ?
|
||||
|
||||
**A :** Cause the best known Chinese WWM BERT is [Chinese-BERT-wwm](https://github.com/ymcui/Chinese-BERT-wwm) by HIT. It works well on so many Chines Task like CLUE (Chinese GLUE).
|
||||
They use LTP, so if we want to fine-tune their model, we need LTP.
|
||||
|
||||
```bash
|
||||
export TRAIN_FILE=/path/to/dataset/wiki.train.raw
|
||||
export LTP_RESOURCE=/path/to/ltp/tokenizer
|
||||
export BERT_RESOURCE=/path/to/bert/tokenizer
|
||||
export SAVE_PATH=/path/to/data/ref.txt
|
||||
|
||||
python chinese_ref.py \
|
||||
--file_name=$TRAIN_FILE \
|
||||
--ltp=$LTP_RESOURCE
|
||||
--bert=$BERT_RESOURCE \
|
||||
--save_path=$SAVE_PATH
|
||||
```
|
||||
Now Chinese Ref is only supported by `LineByLineWithRefDataset` Class, so we need add `line_by_line` flag:
|
||||
|
||||
|
||||
```bash
|
||||
export TRAIN_FILE=/path/to/dataset/wiki.train.raw
|
||||
export TEST_FILE=/path/to/dataset/wiki.test.raw
|
||||
export REF_FILE=/path/to/ref.txt
|
||||
|
||||
python run_language_modeling.py \
|
||||
--output_dir=output \
|
||||
--model_type=roberta \
|
||||
--model_name_or_path=roberta-base \
|
||||
--do_train \
|
||||
--train_data_file=$TRAIN_FILE \
|
||||
--chinese_ref_file=$REF_FILE \
|
||||
--do_eval \
|
||||
--eval_data_file=$TEST_FILE \
|
||||
--mlm \
|
||||
--line_by_line \
|
||||
--wwm
|
||||
```
|
||||
|
||||
### XLNet and permutation language modeling
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
import argparse
|
||||
import json
|
||||
from typing import List
|
||||
|
||||
from ltp import LTP
|
||||
from transformers.tokenization_bert import BertTokenizer
|
||||
|
||||
|
||||
def _is_chinese_char(cp):
|
||||
"""Checks whether CP is the codepoint of a CJK character."""
|
||||
# This defines a "chinese character" as anything in the CJK Unicode block:
|
||||
# https://en.wikipedia.org/wiki/CJK_Unified_Ideographs_(Unicode_block)
|
||||
#
|
||||
# Note that the CJK Unicode block is NOT all Japanese and Korean characters,
|
||||
# despite its name. The modern Korean Hangul alphabet is a different block,
|
||||
# as is Japanese Hiragana and Katakana. Those alphabets are used to write
|
||||
# space-separated words, so they are not treated specially and handled
|
||||
# like the all of the other languages.
|
||||
if (
|
||||
(cp >= 0x4E00 and cp <= 0x9FFF)
|
||||
or (cp >= 0x3400 and cp <= 0x4DBF) #
|
||||
or (cp >= 0x20000 and cp <= 0x2A6DF) #
|
||||
or (cp >= 0x2A700 and cp <= 0x2B73F) #
|
||||
or (cp >= 0x2B740 and cp <= 0x2B81F) #
|
||||
or (cp >= 0x2B820 and cp <= 0x2CEAF) #
|
||||
or (cp >= 0xF900 and cp <= 0xFAFF)
|
||||
or (cp >= 0x2F800 and cp <= 0x2FA1F) #
|
||||
): #
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def is_chinese(word: str):
|
||||
# word like '180' or '身高' or '神'
|
||||
for char in word:
|
||||
char = ord(char)
|
||||
if not _is_chinese_char(char):
|
||||
return 0
|
||||
return 1
|
||||
|
||||
|
||||
def get_chinese_word(tokens: List[str]):
|
||||
word_set = set()
|
||||
|
||||
for token in tokens:
|
||||
chinese_word = len(token) > 1 and is_chinese(token)
|
||||
if chinese_word:
|
||||
word_set.add(token)
|
||||
word_list = list(word_set)
|
||||
return word_list
|
||||
|
||||
|
||||
def add_sub_symbol(bert_tokens: List[str], chinese_word_set: set()):
|
||||
if not chinese_word_set:
|
||||
return bert_tokens
|
||||
max_word_len = max([len(w) for w in chinese_word_set])
|
||||
|
||||
bert_word = bert_tokens
|
||||
start, end = 0, len(bert_word)
|
||||
while start < end:
|
||||
single_word = True
|
||||
if is_chinese(bert_word[start]):
|
||||
l = min(end - start, max_word_len)
|
||||
for i in range(l, 1, -1):
|
||||
whole_word = "".join(bert_word[start : start + i])
|
||||
if whole_word in chinese_word_set:
|
||||
for j in range(start + 1, start + i):
|
||||
bert_word[j] = "##" + bert_word[j]
|
||||
start = start + i
|
||||
single_word = False
|
||||
break
|
||||
if single_word:
|
||||
start += 1
|
||||
return bert_word
|
||||
|
||||
|
||||
def prepare_ref(lines: List[str], ltp_tokenizer: LTP, bert_tokenizer: BertTokenizer):
|
||||
ltp_res = []
|
||||
|
||||
for i in range(0, len(lines), 100):
|
||||
res = ltp_tokenizer.seg(lines[i : i + 100])[0]
|
||||
res = [get_chinese_word(r) for r in res]
|
||||
ltp_res.extend(res)
|
||||
assert len(ltp_res) == len(lines)
|
||||
|
||||
bert_res = []
|
||||
for i in range(0, len(lines), 100):
|
||||
res = bert_tokenizer(lines[i : i + 100], add_special_tokens=True, truncation=True, max_length=512)
|
||||
bert_res.extend(res["input_ids"])
|
||||
assert len(bert_res) == len(lines)
|
||||
|
||||
ref_ids = []
|
||||
for input_ids, chinese_word in zip(bert_res, ltp_res):
|
||||
|
||||
input_tokens = []
|
||||
for id in input_ids:
|
||||
token = bert_tokenizer._convert_id_to_token(id)
|
||||
input_tokens.append(token)
|
||||
input_tokens = add_sub_symbol(input_tokens, chinese_word)
|
||||
ref_id = []
|
||||
# We only save pos of chinese subwords start with ##, which mean is part of a whole word.
|
||||
for i, token in enumerate(input_tokens):
|
||||
if token[:2] == "##":
|
||||
clean_token = token[2:]
|
||||
# save chinese tokens' pos
|
||||
if len(clean_token) == 1 and _is_chinese_char(ord(clean_token)):
|
||||
ref_id.append(i)
|
||||
ref_ids.append(ref_id)
|
||||
|
||||
assert len(ref_ids) == len(bert_res)
|
||||
|
||||
return ref_ids
|
||||
|
||||
|
||||
def main(args):
|
||||
# For Chinese (Ro)Bert, the best result is from : RoBERTa-wwm-ext (https://github.com/ymcui/Chinese-BERT-wwm)
|
||||
# If we want to fine-tune these model, we have to use same tokenizer : LTP (https://github.com/HIT-SCIR/ltp)
|
||||
with open(args.file_name, "r", encoding="utf-8") as f:
|
||||
data = f.readlines()
|
||||
|
||||
ltp_tokenizer = LTP(args.ltp) # faster in GPU device
|
||||
bert_tokenizer = BertTokenizer.from_pretrained(args.bert)
|
||||
|
||||
ref_ids = prepare_ref(data, ltp_tokenizer, bert_tokenizer)
|
||||
|
||||
with open(args.save_path, "w", encoding="utf-8") as f:
|
||||
data = [json.dumps(ref) + "\n" for ref in ref_ids]
|
||||
f.writelines(data)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="prepare_chinese_ref")
|
||||
parser.add_argument(
|
||||
"--file_name",
|
||||
type=str,
|
||||
default="./resources/chinese-demo.txt",
|
||||
help="file need process, same as training data in lm",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--ltp", type=str, default="./resources/ltp", help="resources for LTP tokenizer, usually a path"
|
||||
)
|
||||
parser.add_argument("--bert", type=str, default="./resources/robert", help="resources for Bert tokenizer")
|
||||
parser.add_argument("--save_path", type=str, default="./resources/ref.txt", help="path to save res")
|
||||
|
||||
args = parser.parse_args()
|
||||
main(args)
|
||||
@@ -37,8 +37,10 @@ from transformers import (
|
||||
AutoTokenizer,
|
||||
DataCollatorForLanguageModeling,
|
||||
DataCollatorForPermutationLanguageModeling,
|
||||
DataCollatorForWholeWordMask,
|
||||
HfArgumentParser,
|
||||
LineByLineTextDataset,
|
||||
LineByLineWithRefDataset,
|
||||
PreTrainedTokenizer,
|
||||
TextDataset,
|
||||
Trainer,
|
||||
@@ -101,6 +103,10 @@ class DataTrainingArguments:
|
||||
default=None,
|
||||
metadata={"help": "An optional input evaluation data file to evaluate the perplexity on (a text file)."},
|
||||
)
|
||||
chinese_ref_file: Optional[str] = field(
|
||||
default=None,
|
||||
metadata={"help": "An optional input ref data file for whole word mask in Chinees."},
|
||||
)
|
||||
line_by_line: bool = field(
|
||||
default=False,
|
||||
metadata={"help": "Whether distinct lines of text in the dataset are to be handled as distinct sequences."},
|
||||
@@ -109,6 +115,7 @@ class DataTrainingArguments:
|
||||
mlm: bool = field(
|
||||
default=False, metadata={"help": "Train with masked-language modeling loss instead of language modeling."}
|
||||
)
|
||||
whole_word_mask: bool = field(default=False, metadata={"help": "Whether ot not to use whole word mask."})
|
||||
mlm_probability: float = field(
|
||||
default=0.15, metadata={"help": "Ratio of tokens to mask for masked language modeling loss"}
|
||||
)
|
||||
@@ -143,6 +150,16 @@ def get_dataset(
|
||||
):
|
||||
def _dataset(file_path):
|
||||
if args.line_by_line:
|
||||
if args.chinese_ref_file is not None:
|
||||
if not args.whole_word_mask or not args.mlm:
|
||||
raise ValueError("You need to set world whole masking and mlm to True for Chinese Whole Word Mask")
|
||||
return LineByLineWithRefDataset(
|
||||
tokenizer=tokenizer,
|
||||
file_path=file_path,
|
||||
block_size=args.block_size,
|
||||
ref_path=args.chinese_ref_file,
|
||||
)
|
||||
|
||||
return LineByLineTextDataset(tokenizer=tokenizer, file_path=file_path, block_size=args.block_size)
|
||||
else:
|
||||
return TextDataset(
|
||||
@@ -174,7 +191,6 @@ def main():
|
||||
"Cannot do evaluation without an evaluation data file. Either supply a file to --eval_data_file "
|
||||
"or remove the --do_eval argument."
|
||||
)
|
||||
|
||||
if (
|
||||
os.path.exists(training_args.output_dir)
|
||||
and os.listdir(training_args.output_dir)
|
||||
@@ -270,9 +286,14 @@ def main():
|
||||
max_span_length=data_args.max_span_length,
|
||||
)
|
||||
else:
|
||||
data_collator = DataCollatorForLanguageModeling(
|
||||
tokenizer=tokenizer, mlm=data_args.mlm, mlm_probability=data_args.mlm_probability
|
||||
)
|
||||
if data_args.mlm and data_args.whole_word_mask:
|
||||
data_collator = DataCollatorForWholeWordMask(
|
||||
tokenizer=tokenizer, mlm_probability=data_args.mlm_probability
|
||||
)
|
||||
else:
|
||||
data_collator = DataCollatorForLanguageModeling(
|
||||
tokenizer=tokenizer, mlm=data_args.mlm, mlm_probability=data_args.mlm_probability
|
||||
)
|
||||
|
||||
# Initialize our Trainer
|
||||
trainer = Trainer(
|
||||
|
||||
@@ -170,7 +170,7 @@ class BaseTransformer(pl.LightningModule):
|
||||
self.dataset_size = len(self.test_dataloader().dataset)
|
||||
else:
|
||||
self.train_loader = self.get_dataloader("train", self.hparams.train_batch_size, shuffle=True)
|
||||
self.dataset_size = len(self.train_loader.dataset)
|
||||
self.dataset_size = len(self.train_dataloader().dataset)
|
||||
|
||||
def get_dataloader(self, type_path: str, batch_size: int, shuffle: bool = False):
|
||||
raise NotImplementedError("You must implement this for your task")
|
||||
|
||||
@@ -187,7 +187,7 @@ def train(args, train_dataset, model, tokenizer):
|
||||
"end_positions": batch[4],
|
||||
}
|
||||
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart"]:
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart", "longformer"]:
|
||||
del inputs["token_type_ids"]
|
||||
|
||||
if args.model_type in ["xlnet", "xlm"]:
|
||||
@@ -300,7 +300,7 @@ def evaluate(args, model, tokenizer, prefix=""):
|
||||
"token_type_ids": batch[2],
|
||||
}
|
||||
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart"]:
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart", "longformer"]:
|
||||
del inputs["token_type_ids"]
|
||||
|
||||
feature_indices = batch[3]
|
||||
|
||||
@@ -35,9 +35,10 @@ def split_documents(documents: dict) -> dict:
|
||||
"""Split documents into passages"""
|
||||
titles, texts = [], []
|
||||
for title, text in zip(documents["title"], documents["text"]):
|
||||
for passage in split_text(text):
|
||||
titles.append(title)
|
||||
texts.append(passage)
|
||||
if text is not None:
|
||||
for passage in split_text(text):
|
||||
titles.append(title if title is not None else "")
|
||||
texts.append(passage)
|
||||
return {"title": titles, "text": texts}
|
||||
|
||||
|
||||
|
||||
@@ -15,7 +15,8 @@ For `bertabs` instructions, see [`bertabs/README.md`](bertabs/README.md).
|
||||
|
||||
## Datasets
|
||||
|
||||
#### XSUM:
|
||||
#### XSUM
|
||||
|
||||
```bash
|
||||
cd examples/seq2seq
|
||||
wget https://cdn-datasets.huggingface.co/summarization/xsum.tar.gz
|
||||
@@ -26,6 +27,7 @@ this should make a directory called `xsum/` with files like `test.source`.
|
||||
To use your own data, copy that files format. Each article to be summarized is on its own line.
|
||||
|
||||
#### CNN/DailyMail
|
||||
|
||||
```bash
|
||||
cd examples/seq2seq
|
||||
wget https://cdn-datasets.huggingface.co/summarization/cnn_dm_v2.tgz
|
||||
@@ -35,7 +37,8 @@ export CNN_DIR=${PWD}/cnn_dm
|
||||
```
|
||||
this should make a directory called `cnn_dm/` with 6 files.
|
||||
|
||||
#### WMT16 English-Romanian Translation Data:
|
||||
#### WMT16 English-Romanian Translation Data
|
||||
|
||||
download with this command:
|
||||
```bash
|
||||
wget https://cdn-datasets.huggingface.co/translation/wmt_en_ro.tar.gz
|
||||
@@ -44,13 +47,25 @@ export ENRO_DIR=${PWD}/wmt_en_ro
|
||||
```
|
||||
this should make a directory called `wmt_en_ro/` with 6 files.
|
||||
|
||||
#### WMT English-German:
|
||||
#### WMT English-German
|
||||
|
||||
```bash
|
||||
wget https://cdn-datasets.huggingface.co/translation/wmt_en_de.tgz
|
||||
tar -xzvf wmt_en_de.tgz
|
||||
export DATA_DIR=${PWD}/wmt_en_de
|
||||
```
|
||||
|
||||
#### FSMT datasets (wmt)
|
||||
|
||||
Refer to the scripts starting with `eval_` under:
|
||||
https://github.com/huggingface/transformers/tree/master/scripts/fsmt
|
||||
|
||||
#### Pegasus (multiple datasets)
|
||||
|
||||
Multiple eval datasets are available for download from:
|
||||
https://github.com/stas00/porting/tree/master/datasets/pegasus
|
||||
|
||||
|
||||
#### Private Data
|
||||
|
||||
If you are using your own data, it must be formatted as one directory with 6 files:
|
||||
@@ -64,7 +79,6 @@ test.target
|
||||
```
|
||||
The `.source` files are the input, the `.target` files are the desired output.
|
||||
|
||||
|
||||
### Tips and Tricks
|
||||
|
||||
General Tips:
|
||||
|
||||
@@ -17,7 +17,7 @@ from finetune import main as ft_main
|
||||
from make_student import create_student_by_copying_alternating_layers, get_layers_to_supervise
|
||||
from transformers import AutoModelForSeq2SeqLM, MBartTokenizer, T5ForConditionalGeneration
|
||||
from transformers.modeling_bart import shift_tokens_right
|
||||
from utils import calculate_bleu, freeze_params, label_smoothed_nll_loss, use_task_specific_params
|
||||
from utils import calculate_bleu, check_output_dir, freeze_params, label_smoothed_nll_loss, use_task_specific_params
|
||||
|
||||
|
||||
# need the parent dir module
|
||||
@@ -266,8 +266,7 @@ def create_module(args):
|
||||
|
||||
def distill_main(args):
|
||||
Path(args.output_dir).mkdir(exist_ok=True)
|
||||
if len(os.listdir(args.output_dir)) > 3 and args.do_train:
|
||||
raise ValueError("Output directory ({}) already exists and is not empty.".format(args.output_dir))
|
||||
check_output_dir(args, expected_items=3)
|
||||
|
||||
model = create_module(args)
|
||||
return ft_main(args, model=model)
|
||||
|
||||
@@ -25,6 +25,7 @@ from utils import (
|
||||
assert_all_frozen,
|
||||
calculate_bleu,
|
||||
calculate_rouge,
|
||||
check_output_dir,
|
||||
flatten_list,
|
||||
freeze_embeds,
|
||||
freeze_params,
|
||||
@@ -329,6 +330,7 @@ class SummarizationModule(BaseTransformer):
|
||||
parser.add_argument("--freeze_encoder", action="store_true")
|
||||
parser.add_argument("--freeze_embeds", action="store_true")
|
||||
parser.add_argument("--sortish_sampler", action="store_true", default=False)
|
||||
parser.add_argument("--overwrite_output_dir", action="store_true", default=False)
|
||||
parser.add_argument("--max_tokens_per_batch", type=int, default=None)
|
||||
parser.add_argument("--logger_name", type=str, choices=["default", "wandb", "wandb_shared"], default="default")
|
||||
parser.add_argument("--n_train", type=int, default=-1, required=False, help="# examples. -1 means use all.")
|
||||
@@ -373,8 +375,8 @@ class TranslationModule(SummarizationModule):
|
||||
|
||||
def main(args, model=None) -> SummarizationModule:
|
||||
Path(args.output_dir).mkdir(exist_ok=True)
|
||||
if len(os.listdir(args.output_dir)) > 3 and args.do_train:
|
||||
raise ValueError("Output directory ({}) already exists and is not empty.".format(args.output_dir))
|
||||
check_output_dir(args, expected_items=3)
|
||||
|
||||
if model is None:
|
||||
if "summarization" in args.task:
|
||||
model: SummarizationModule = SummarizationModule(args)
|
||||
|
||||
@@ -16,11 +16,11 @@ from transformers import (
|
||||
)
|
||||
from transformers.trainer_utils import EvaluationStrategy
|
||||
from utils import (
|
||||
LegacySeq2SeqDataset,
|
||||
Seq2SeqDataCollator,
|
||||
Seq2SeqDataset,
|
||||
assert_all_frozen,
|
||||
build_compute_metrics_fn,
|
||||
check_output_dir,
|
||||
freeze_embeds,
|
||||
freeze_params,
|
||||
lmap,
|
||||
@@ -137,6 +137,10 @@ class DataTrainingArguments:
|
||||
src_lang: Optional[str] = field(default=None, metadata={"help": "Source language id for translation."})
|
||||
tgt_lang: Optional[str] = field(default=None, metadata={"help": "Target language id for translation."})
|
||||
eval_beams: Optional[int] = field(default=None, metadata={"help": "# num_beams to use for evaluation."})
|
||||
ignore_pad_token_for_loss: bool = field(
|
||||
default=True,
|
||||
metadata={"help": "If only pad tokens should be ignored. This assumes that `config.pad_token_id` is defined."},
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
@@ -153,15 +157,7 @@ def main():
|
||||
else:
|
||||
model_args, data_args, training_args = parser.parse_args_into_dataclasses()
|
||||
|
||||
if (
|
||||
os.path.exists(training_args.output_dir)
|
||||
and os.listdir(training_args.output_dir)
|
||||
and training_args.do_train
|
||||
and not training_args.overwrite_output_dir
|
||||
):
|
||||
raise ValueError(
|
||||
f"Output directory ({training_args.output_dir}) already exists and is not empty. Use --overwrite_output_dir to overcome."
|
||||
)
|
||||
check_output_dir(training_args)
|
||||
|
||||
# Setup logging
|
||||
logging.basicConfig(
|
||||
@@ -230,7 +226,7 @@ def main():
|
||||
freeze_params(model.get_encoder())
|
||||
assert_all_frozen(model.get_encoder())
|
||||
|
||||
dataset_class = Seq2SeqDataset if hasattr(tokenizer, "prepare_seq2seq_batch") else LegacySeq2SeqDataset
|
||||
dataset_class = Seq2SeqDataset
|
||||
|
||||
# Get datasets
|
||||
train_dataset = (
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
import logging
|
||||
import copy
|
||||
from typing import Any, Dict, Optional, Tuple, Union
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
from torch.utils.data import DistributedSampler, RandomSampler
|
||||
|
||||
from transformers import Trainer
|
||||
from transformers import PreTrainedModel, Trainer, logging
|
||||
from transformers.configuration_fsmt import FSMTConfig
|
||||
from transformers.file_utils import is_torch_tpu_available
|
||||
from transformers.optimization import (
|
||||
@@ -27,7 +27,7 @@ except ImportError:
|
||||
from utils import label_smoothed_nll_loss
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
arg_to_scheduler = {
|
||||
"linear": get_linear_schedule_with_warmup,
|
||||
@@ -41,13 +41,25 @@ arg_to_scheduler_choices = sorted(arg_to_scheduler.keys())
|
||||
|
||||
|
||||
class Seq2SeqTrainer(Trainer):
|
||||
def __init__(self, config, data_args, *args, **kwargs):
|
||||
def __init__(self, config=None, data_args=None, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
self.config = config
|
||||
|
||||
if config is None:
|
||||
assert isinstance(
|
||||
self.model, PreTrainedModel
|
||||
), f"If no `config` is passed the model to be trained has to be of type `PreTrainedModel`, but is {self.model.__class__}"
|
||||
self.config = self._actual_model(self.model).config
|
||||
else:
|
||||
self.config = config
|
||||
|
||||
self.data_args = data_args
|
||||
self.max_gen_length = data_args.val_max_target_length
|
||||
self.vocab_size = self.config.tgt_vocab_size if isinstance(self.config, FSMTConfig) else self.config.vocab_size
|
||||
|
||||
if self.args.label_smoothing != 0 or (self.data_args is not None and self.data_args.ignore_pad_token_for_loss):
|
||||
assert (
|
||||
self.config.pad_token_id is not None
|
||||
), "Make sure that `config.pad_token_id` is correcly defined when ignoring `pad_token` for loss calculation or doing label smoothing."
|
||||
|
||||
def create_optimizer_and_scheduler(self, num_training_steps: int):
|
||||
"""
|
||||
Setup the optimizer and the learning rate scheduler.
|
||||
@@ -114,23 +126,31 @@ class Seq2SeqTrainer(Trainer):
|
||||
else DistributedSampler(self.train_dataset)
|
||||
)
|
||||
|
||||
def compute_loss(self, model, inputs):
|
||||
labels = inputs.pop("labels")
|
||||
outputs = model(**inputs, use_cache=False)
|
||||
logits = outputs[0]
|
||||
return self._compute_loss(logits, labels)
|
||||
|
||||
def _compute_loss(self, logits, labels):
|
||||
def _compute_loss(self, model, inputs):
|
||||
inputs = copy.deepcopy(inputs)
|
||||
if self.args.label_smoothing == 0:
|
||||
# Same behavior as modeling_bart.py
|
||||
loss_fct = torch.nn.CrossEntropyLoss(ignore_index=self.config.pad_token_id)
|
||||
assert logits.shape[-1] == self.vocab_size
|
||||
loss = loss_fct(logits.view(-1, logits.shape[-1]), labels.view(-1))
|
||||
if self.data_args is not None and self.data_args.ignore_pad_token_for_loss:
|
||||
# force training to ignore pad token
|
||||
labels = inputs.pop("labels")
|
||||
logits = model(**inputs, use_cache=False)[0]
|
||||
|
||||
loss_fct = torch.nn.CrossEntropyLoss(ignore_index=self.config.pad_token_id)
|
||||
loss = loss_fct(logits.view(-1, logits.shape[-1]), labels.view(-1))
|
||||
else:
|
||||
# compute usual loss via models
|
||||
loss, logits = model(**inputs, use_cache=False)[:2]
|
||||
else:
|
||||
# compute label smoothed loss
|
||||
labels = inputs.pop("labels")
|
||||
logits = model(**inputs, use_cache=False)[0]
|
||||
lprobs = torch.nn.functional.log_softmax(logits, dim=-1)
|
||||
loss, nll_loss = label_smoothed_nll_loss(
|
||||
loss, _ = label_smoothed_nll_loss(
|
||||
lprobs, labels, self.args.label_smoothing, ignore_index=self.config.pad_token_id
|
||||
)
|
||||
return loss, logits
|
||||
|
||||
def compute_loss(self, model, inputs):
|
||||
loss, _ = self._compute_loss(model, inputs)
|
||||
return loss
|
||||
|
||||
def prediction_step(
|
||||
@@ -158,31 +178,37 @@ class Seq2SeqTrainer(Trainer):
|
||||
"""
|
||||
inputs = self._prepare_inputs(inputs)
|
||||
|
||||
if self.args.predict_with_generate and not self.args.prediction_loss_only:
|
||||
gen_kwargs = {
|
||||
"max_length": self.data_args.val_max_target_length
|
||||
if self.data_args is not None
|
||||
else self.config.max_length,
|
||||
"num_beams": self.data_args.eval_beams if self.data_args is not None else self.config.num_beams,
|
||||
}
|
||||
generated_tokens = model.generate(
|
||||
inputs["input_ids"],
|
||||
attention_mask=inputs["attention_mask"],
|
||||
**gen_kwargs,
|
||||
)
|
||||
# in case the batch is shorter than max length, the output should be padded
|
||||
if self.config.pad_token_id is not None:
|
||||
generated_tokens = self._pad_tensors_to_max_len(generated_tokens, gen_kwargs["max_length"])
|
||||
|
||||
# compute loss on predict data
|
||||
with torch.no_grad():
|
||||
if self.args.predict_with_generate and not self.args.prediction_loss_only:
|
||||
generated_tokens = model.generate(
|
||||
inputs["input_ids"],
|
||||
attention_mask=inputs["attention_mask"],
|
||||
use_cache=True,
|
||||
num_beams=self.data_args.eval_beams,
|
||||
max_length=self.max_gen_length,
|
||||
)
|
||||
# in case the batch is shorter than max length, the output should be padded
|
||||
generated_tokens = self._pad_tensors_to_max_len(generated_tokens, self.max_gen_length)
|
||||
loss, logits = self._compute_loss(model, inputs)
|
||||
|
||||
labels_out = inputs.get("labels")
|
||||
# Call forward again to get loss # TODO: avoidable?
|
||||
outputs = model(**inputs, use_cache=False)
|
||||
loss = self._compute_loss(outputs[1], labels_out)
|
||||
loss = loss.mean().detach()
|
||||
if self.args.prediction_loss_only:
|
||||
return (loss, None, None)
|
||||
loss = loss.mean().detach()
|
||||
if self.args.prediction_loss_only:
|
||||
return (loss, None, None)
|
||||
|
||||
logits = generated_tokens if self.args.predict_with_generate else outputs[1]
|
||||
logits = generated_tokens if self.args.predict_with_generate else logits
|
||||
|
||||
labels_out = labels_out.detach()
|
||||
labels = self._pad_tensors_to_max_len(labels_out, self.max_gen_length)
|
||||
return (loss, logits.detach(), labels)
|
||||
labels = inputs["labels"]
|
||||
if self.config.pad_token_id is not None:
|
||||
labels = self._pad_tensors_to_max_len(labels, self.config.max_length)
|
||||
|
||||
return (loss, logits, labels)
|
||||
|
||||
def _pad_tensors_to_max_len(self, tensor, max_length):
|
||||
padded_tensor = self.config.pad_token_id * torch.ones(
|
||||
|
||||
@@ -1,15 +1,25 @@
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from transformers import BertTokenizer, EncoderDecoderModel, is_torch_available
|
||||
from transformers.file_utils import is_datasets_available
|
||||
from transformers.testing_utils import TestCasePlus, slow
|
||||
from transformers.trainer_callback import TrainerState
|
||||
from transformers.trainer_utils import set_seed
|
||||
|
||||
from .finetune_trainer import main
|
||||
from .finetune_trainer import Seq2SeqTrainingArguments, main
|
||||
from .seq2seq_trainer import Seq2SeqTrainer
|
||||
from .test_seq2seq_examples import MBART_TINY
|
||||
from .utils import execute_async_std
|
||||
|
||||
|
||||
if is_torch_available():
|
||||
import torch
|
||||
|
||||
set_seed(42)
|
||||
MARIAN_MODEL = "sshleifer/student_marian_en_ro_6_1"
|
||||
|
||||
@@ -25,7 +35,7 @@ class TestFinetuneTrainer(TestCasePlus):
|
||||
@slow
|
||||
def test_finetune_trainer_slow(self):
|
||||
# There is a missing call to __init__process_group somewhere
|
||||
output_dir = self.run_trainer(eval_steps=2, max_len="128", model_name=MARIAN_MODEL, num_train_epochs=3)
|
||||
output_dir = self.run_trainer(eval_steps=2, max_len="128", model_name=MARIAN_MODEL, num_train_epochs=10)
|
||||
|
||||
# Check metrics
|
||||
logs = TrainerState.load_from_json(os.path.join(output_dir, "trainer_state.json")).log_history
|
||||
@@ -42,7 +52,120 @@ class TestFinetuneTrainer(TestCasePlus):
|
||||
assert "test_generations.txt" in contents
|
||||
assert "test_results.json" in contents
|
||||
|
||||
@slow
|
||||
def test_finetune_bert2bert(self):
|
||||
if not is_datasets_available():
|
||||
return
|
||||
|
||||
import datasets
|
||||
|
||||
bert2bert = EncoderDecoderModel.from_encoder_decoder_pretrained("prajjwal1/bert-tiny", "prajjwal1/bert-tiny")
|
||||
tokenizer = BertTokenizer.from_pretrained("bert-base-uncased")
|
||||
|
||||
bert2bert.config.vocab_size = bert2bert.config.encoder.vocab_size
|
||||
bert2bert.config.decoder_start_token_id = tokenizer.cls_token_id
|
||||
|
||||
train_dataset = datasets.load_dataset("cnn_dailymail", "3.0.0", split="train[:1%]")
|
||||
val_dataset = datasets.load_dataset("cnn_dailymail", "3.0.0", split="validation[:1%]")
|
||||
|
||||
train_dataset = train_dataset.select(range(32))
|
||||
val_dataset = val_dataset.select(range(16))
|
||||
|
||||
rouge = datasets.load_metric("rouge")
|
||||
|
||||
batch_size = 4
|
||||
|
||||
def _map_to_encoder_decoder_inputs(batch):
|
||||
# Tokenizer will automatically set [BOS] <text> [EOS]
|
||||
inputs = tokenizer(batch["article"], padding="max_length", truncation=True, max_length=512)
|
||||
outputs = tokenizer(batch["highlights"], padding="max_length", truncation=True, max_length=128)
|
||||
batch["input_ids"] = inputs.input_ids
|
||||
batch["attention_mask"] = inputs.attention_mask
|
||||
|
||||
batch["decoder_input_ids"] = outputs.input_ids
|
||||
batch["labels"] = outputs.input_ids.copy()
|
||||
batch["labels"] = [
|
||||
[-100 if token == tokenizer.pad_token_id else token for token in labels] for labels in batch["labels"]
|
||||
]
|
||||
batch["decoder_attention_mask"] = outputs.attention_mask
|
||||
|
||||
assert all([len(x) == 512 for x in inputs.input_ids])
|
||||
assert all([len(x) == 128 for x in outputs.input_ids])
|
||||
|
||||
return batch
|
||||
|
||||
def _compute_metrics(pred):
|
||||
labels_ids = pred.label_ids
|
||||
pred_ids = pred.predictions
|
||||
|
||||
# all unnecessary tokens are removed
|
||||
pred_str = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)
|
||||
label_str = tokenizer.batch_decode(labels_ids, skip_special_tokens=True)
|
||||
|
||||
rouge_output = rouge.compute(predictions=pred_str, references=label_str, rouge_types=["rouge2"])[
|
||||
"rouge2"
|
||||
].mid
|
||||
|
||||
return {
|
||||
"rouge2_precision": round(rouge_output.precision, 4),
|
||||
"rouge2_recall": round(rouge_output.recall, 4),
|
||||
"rouge2_fmeasure": round(rouge_output.fmeasure, 4),
|
||||
}
|
||||
|
||||
# map train dataset
|
||||
train_dataset = train_dataset.map(
|
||||
_map_to_encoder_decoder_inputs,
|
||||
batched=True,
|
||||
batch_size=batch_size,
|
||||
remove_columns=["article", "highlights"],
|
||||
)
|
||||
train_dataset.set_format(
|
||||
type="torch",
|
||||
columns=["input_ids", "attention_mask", "decoder_input_ids", "decoder_attention_mask", "labels"],
|
||||
)
|
||||
|
||||
# same for validation dataset
|
||||
val_dataset = val_dataset.map(
|
||||
_map_to_encoder_decoder_inputs,
|
||||
batched=True,
|
||||
batch_size=batch_size,
|
||||
remove_columns=["article", "highlights"],
|
||||
)
|
||||
val_dataset.set_format(
|
||||
type="torch",
|
||||
columns=["input_ids", "attention_mask", "decoder_input_ids", "decoder_attention_mask", "labels"],
|
||||
)
|
||||
|
||||
output_dir = self.get_auto_remove_tmp_dir()
|
||||
|
||||
training_args = Seq2SeqTrainingArguments(
|
||||
output_dir=output_dir,
|
||||
per_device_train_batch_size=batch_size,
|
||||
per_device_eval_batch_size=batch_size,
|
||||
predict_with_generate=True,
|
||||
evaluate_during_training=True,
|
||||
do_train=True,
|
||||
do_eval=True,
|
||||
warmup_steps=0,
|
||||
eval_steps=2,
|
||||
logging_steps=2,
|
||||
)
|
||||
|
||||
# instantiate trainer
|
||||
trainer = Seq2SeqTrainer(
|
||||
model=bert2bert,
|
||||
args=training_args,
|
||||
compute_metrics=_compute_metrics,
|
||||
train_dataset=train_dataset,
|
||||
eval_dataset=val_dataset,
|
||||
)
|
||||
|
||||
# start training
|
||||
trainer.train()
|
||||
|
||||
def run_trainer(self, eval_steps: int, max_len: str, model_name: str, num_train_epochs: int):
|
||||
|
||||
# XXX: remove hardcoded path
|
||||
data_dir = "examples/seq2seq/test_data/wmt_en_ro"
|
||||
output_dir = self.get_auto_remove_tmp_dir()
|
||||
argv = f"""
|
||||
@@ -77,8 +200,34 @@ class TestFinetuneTrainer(TestCasePlus):
|
||||
""".split()
|
||||
# --eval_beams 2
|
||||
|
||||
testargs = ["finetune_trainer.py"] + argv
|
||||
with patch.object(sys, "argv", testargs):
|
||||
main()
|
||||
n_gpu = torch.cuda.device_count()
|
||||
if n_gpu > 1:
|
||||
|
||||
path = Path(__file__).resolve()
|
||||
cur_path = path.parents[0]
|
||||
|
||||
path = Path(__file__).resolve()
|
||||
examples_path = path.parents[1]
|
||||
src_path = f"{path.parents[2]}/src"
|
||||
env = os.environ.copy()
|
||||
env["PYTHONPATH"] = f"{examples_path}:{src_path}:{env.get('PYTHONPATH', '')}"
|
||||
|
||||
distributed_args = (
|
||||
f"-m torch.distributed.launch --nproc_per_node={n_gpu} {cur_path}/finetune_trainer.py".split()
|
||||
)
|
||||
cmd = [sys.executable] + distributed_args + argv
|
||||
|
||||
print("\nRunning: ", " ".join(cmd))
|
||||
|
||||
result = execute_async_std(cmd, env=env, stdin=None, timeout=180, quiet=False, echo=False)
|
||||
|
||||
assert result.stdout, "produced no output"
|
||||
if result.returncode > 0:
|
||||
pytest.fail(f"failed with returncode {result.returncode}")
|
||||
else:
|
||||
# 0 or 1 gpu
|
||||
testargs = ["finetune_trainer.py"] + argv
|
||||
with patch.object(sys, "argv", testargs):
|
||||
main()
|
||||
|
||||
return output_dir
|
||||
|
||||
@@ -0,0 +1,199 @@
|
||||
# as due to their complexity multi-gpu tests could impact other tests, and to aid debug we have those in a separate module.
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from transformers import is_torch_available
|
||||
from transformers.testing_utils import TestCasePlus, require_torch_multigpu
|
||||
|
||||
from .utils import execute_async_std, load_json
|
||||
|
||||
|
||||
if is_torch_available():
|
||||
import torch
|
||||
|
||||
|
||||
logging.basicConfig(level=logging.DEBUG)
|
||||
|
||||
logger = logging.getLogger()
|
||||
CUDA_AVAILABLE = torch.cuda.is_available()
|
||||
CHEAP_ARGS = {
|
||||
"max_tokens_per_batch": None,
|
||||
"supervise_forward": True,
|
||||
"normalize_hidden": True,
|
||||
"label_smoothing": 0.2,
|
||||
"eval_max_gen_length": None,
|
||||
"eval_beams": 1,
|
||||
"val_metric": "loss",
|
||||
"save_top_k": 1,
|
||||
"adafactor": True,
|
||||
"early_stopping_patience": 2,
|
||||
"logger_name": "default",
|
||||
"length_penalty": 0.5,
|
||||
"cache_dir": "",
|
||||
"task": "summarization",
|
||||
"num_workers": 2,
|
||||
"alpha_hid": 0,
|
||||
"freeze_embeds": True,
|
||||
"enc_only": False,
|
||||
"tgt_suffix": "",
|
||||
"resume_from_checkpoint": None,
|
||||
"sortish_sampler": True,
|
||||
"student_decoder_layers": 1,
|
||||
"val_check_interval": 1.0,
|
||||
"output_dir": "",
|
||||
"fp16": False, # TODO(SS): set this to CUDA_AVAILABLE if ci installs apex or start using native amp
|
||||
"no_teacher": False,
|
||||
"fp16_opt_level": "O1",
|
||||
"gpus": 1 if CUDA_AVAILABLE else 0,
|
||||
"n_tpu_cores": 0,
|
||||
"max_grad_norm": 1.0,
|
||||
"do_train": True,
|
||||
"do_predict": True,
|
||||
"accumulate_grad_batches": 1,
|
||||
"server_ip": "",
|
||||
"server_port": "",
|
||||
"seed": 42,
|
||||
"model_name_or_path": "sshleifer/bart-tiny-random",
|
||||
"config_name": "",
|
||||
"tokenizer_name": "facebook/bart-large",
|
||||
"do_lower_case": False,
|
||||
"learning_rate": 0.3,
|
||||
"lr_scheduler": "linear",
|
||||
"weight_decay": 0.0,
|
||||
"adam_epsilon": 1e-08,
|
||||
"warmup_steps": 0,
|
||||
"max_epochs": 1,
|
||||
"train_batch_size": 2,
|
||||
"eval_batch_size": 2,
|
||||
"max_source_length": 12,
|
||||
"max_target_length": 12,
|
||||
"val_max_target_length": 12,
|
||||
"test_max_target_length": 12,
|
||||
"fast_dev_run": False,
|
||||
"no_cache": False,
|
||||
"n_train": -1,
|
||||
"n_val": -1,
|
||||
"n_test": -1,
|
||||
"student_encoder_layers": 1,
|
||||
"freeze_encoder": False,
|
||||
"auto_scale_batch_size": False,
|
||||
}
|
||||
|
||||
|
||||
def _dump_articles(path: Path, articles: list):
|
||||
content = "\n".join(articles)
|
||||
Path(path).open("w").writelines(content)
|
||||
|
||||
|
||||
ARTICLES = [" Sam ate lunch today.", "Sams lunch ingredients."]
|
||||
SUMMARIES = ["A very interesting story about what I ate for lunch.", "Avocado, celery, turkey, coffee"]
|
||||
T5_TINY = "patrickvonplaten/t5-tiny-random"
|
||||
BART_TINY = "sshleifer/bart-tiny-random"
|
||||
MBART_TINY = "sshleifer/tiny-mbart"
|
||||
MARIAN_TINY = "sshleifer/tiny-marian-en-de"
|
||||
|
||||
|
||||
stream_handler = logging.StreamHandler(sys.stdout)
|
||||
logger.addHandler(stream_handler)
|
||||
logging.disable(logging.CRITICAL) # remove noisy download output from tracebacks
|
||||
|
||||
|
||||
def make_test_data_dir(tmp_dir):
|
||||
for split in ["train", "val", "test"]:
|
||||
_dump_articles(os.path.join(tmp_dir, f"{split}.source"), ARTICLES)
|
||||
_dump_articles(os.path.join(tmp_dir, f"{split}.target"), SUMMARIES)
|
||||
return tmp_dir
|
||||
|
||||
|
||||
class TestSummarizationDistillerMultiGPU(TestCasePlus):
|
||||
@classmethod
|
||||
def setUpClass(cls):
|
||||
logging.disable(logging.CRITICAL) # remove noisy download output from tracebacks
|
||||
return cls
|
||||
|
||||
@require_torch_multigpu
|
||||
def test_multigpu(self):
|
||||
|
||||
updates = dict(
|
||||
no_teacher=True,
|
||||
freeze_encoder=True,
|
||||
gpus=2,
|
||||
overwrite_output_dir=True,
|
||||
sortish_sampler=True,
|
||||
)
|
||||
self._test_distiller_cli_fork(updates, check_contents=False)
|
||||
|
||||
def _test_distiller_cli_fork(self, updates, check_contents=True):
|
||||
default_updates = dict(
|
||||
label_smoothing=0.0,
|
||||
early_stopping_patience=-1,
|
||||
train_batch_size=1,
|
||||
eval_batch_size=2,
|
||||
max_epochs=2,
|
||||
alpha_mlm=0.2,
|
||||
alpha_ce=0.8,
|
||||
do_predict=True,
|
||||
model_name_or_path="sshleifer/tinier_bart",
|
||||
teacher=CHEAP_ARGS["model_name_or_path"],
|
||||
val_check_interval=0.5,
|
||||
)
|
||||
default_updates.update(updates)
|
||||
args_d: dict = CHEAP_ARGS.copy()
|
||||
tmp_dir = make_test_data_dir(tmp_dir=self.get_auto_remove_tmp_dir())
|
||||
output_dir = self.get_auto_remove_tmp_dir()
|
||||
args_d.update(data_dir=tmp_dir, output_dir=output_dir, **default_updates)
|
||||
|
||||
def convert(k, v):
|
||||
if k in ["tgt_suffix", "server_ip", "server_port", "out", "n_tpu_cores"]:
|
||||
return ""
|
||||
if v is False or v is None:
|
||||
return ""
|
||||
if v is True: # or len(str(v))==0:
|
||||
return f"--{k}"
|
||||
return f"--{k}={v}"
|
||||
|
||||
path = Path(__file__).resolve()
|
||||
cur_path = path.parents[0]
|
||||
examples_path = path.parents[1]
|
||||
src_path = f"{path.parents[2]}/src"
|
||||
env = os.environ.copy()
|
||||
env["PYTHONPATH"] = f"{examples_path}:{src_path}:{env.get('PYTHONPATH', '')}"
|
||||
|
||||
cli_args = [x for x in (convert(k, v) for k, v in args_d.items()) if len(x)]
|
||||
cmd = [sys.executable, f"{cur_path}/distillation.py"] + cli_args
|
||||
|
||||
print("\nRunning: ", " ".join(cmd))
|
||||
|
||||
result = execute_async_std(cmd, env=env, stdin=None, timeout=180, quiet=False, echo=False)
|
||||
|
||||
assert result.stdout, "produced no output"
|
||||
if result.returncode > 0:
|
||||
pytest.fail(f"failed with returncode {result.returncode}")
|
||||
|
||||
contents = os.listdir(output_dir)
|
||||
contents = {os.path.basename(p) for p in contents}
|
||||
ckpt_files = [p for p in contents if p.endswith("ckpt")]
|
||||
assert len(ckpt_files) > 0
|
||||
|
||||
self.assertIn("test_generations.txt", contents)
|
||||
self.assertIn("test_results.txt", contents)
|
||||
|
||||
# get the following from the module, (we don't have access to `model` here)
|
||||
metrics_save_path = os.path.join(output_dir, "metrics.json")
|
||||
val_metric = "rouge2"
|
||||
|
||||
metrics = load_json(metrics_save_path)
|
||||
# {'test': [{'test_avg_loss': 10.63731575012207, 'test_avg_rouge1': 0.0, 'test_avg_rouge2': 0.0, 'test_avg_rougeL': 0.0, 'test_avg_gen_time': 0.1822289228439331, 'test_avg_gen_len': 142.0, 'step_count': 1}]}
|
||||
print(metrics)
|
||||
last_step_stats = metrics["val"][-1]
|
||||
self.assertGreaterEqual(last_step_stats["val_avg_gen_time"], 0.01)
|
||||
self.assertGreaterEqual(1.0, last_step_stats["val_avg_gen_time"])
|
||||
self.assertIsInstance(last_step_stats[f"val_avg_{val_metric}"], float)
|
||||
self.assertEqual(len(metrics["test"]), 1)
|
||||
desired_n_evals = int(args_d["max_epochs"] * (1 / args_d["val_check_interval"]) / 2 + 1)
|
||||
self.assertEqual(len(metrics["val"]), desired_n_evals)
|
||||
@@ -5,6 +5,7 @@ import math
|
||||
import os
|
||||
import pickle
|
||||
import socket
|
||||
import sys
|
||||
from logging import getLogger
|
||||
from pathlib import Path
|
||||
from typing import Callable, Dict, Iterable, List, Tuple, Union
|
||||
@@ -619,3 +620,95 @@ def chunks(lst, n):
|
||||
"""Yield successive n-sized chunks from lst."""
|
||||
for i in range(0, len(lst), n):
|
||||
yield lst[i : i + n]
|
||||
|
||||
|
||||
def check_output_dir(args, expected_items=0):
|
||||
"""
|
||||
Checks whether to bail out if output_dir already exists and has more than expected_items in it
|
||||
|
||||
`args`: needs to have the following attributes of `args`:
|
||||
- output_dir
|
||||
- do_train
|
||||
- overwrite_output_dir
|
||||
|
||||
`expected_items`: normally 0 (default) - i.e. empty dir, but in some cases a few files are expected (e.g. recovery from OOM)
|
||||
"""
|
||||
if (
|
||||
os.path.exists(args.output_dir)
|
||||
and len(os.listdir(args.output_dir)) > expected_items
|
||||
and args.do_train
|
||||
and not args.overwrite_output_dir
|
||||
):
|
||||
raise ValueError(
|
||||
f"Output directory ({args.output_dir}) already exists and "
|
||||
"has {len(os.listdir(args.output_dir))} items in it (expected {expected_items} items). "
|
||||
"Use --overwrite_output_dir to overcome."
|
||||
)
|
||||
|
||||
|
||||
# the following code deals with async io between processes
|
||||
|
||||
# adapted from https://stackoverflow.com/a/59041913/9201239
|
||||
import asyncio # noqa
|
||||
|
||||
|
||||
class _RunOutput:
|
||||
def __init__(self, returncode, stdout, stderr):
|
||||
self.returncode = returncode
|
||||
self.stdout = stdout
|
||||
self.stderr = stderr
|
||||
|
||||
|
||||
async def _read_stream(stream, callback):
|
||||
while True:
|
||||
line = await stream.readline()
|
||||
if line:
|
||||
callback(line)
|
||||
else:
|
||||
break
|
||||
|
||||
|
||||
async def _stream_subprocess(cmd, env=None, stdin=None, timeout=None, quiet=False, echo=False) -> _RunOutput:
|
||||
if echo:
|
||||
print(cmd)
|
||||
|
||||
p = await asyncio.create_subprocess_exec(
|
||||
cmd[0],
|
||||
*cmd[1:],
|
||||
stdin=stdin,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
env=env,
|
||||
)
|
||||
out = []
|
||||
err = []
|
||||
|
||||
def tee(line, sink, pipe, label=""):
|
||||
line = line.decode("utf-8").rstrip()
|
||||
sink.append(line)
|
||||
if not quiet:
|
||||
print(label, line, file=pipe)
|
||||
|
||||
await asyncio.wait(
|
||||
[
|
||||
_read_stream(p.stdout, lambda l: tee(l, out, sys.stdout)),
|
||||
_read_stream(p.stderr, lambda l: tee(l, err, sys.stderr, label="stderr:")),
|
||||
],
|
||||
timeout=timeout,
|
||||
)
|
||||
|
||||
# XXX: warning for a possible deadlock when using `wait` with huge amounts of data in the pipe
|
||||
# https://docs.python.org/3/library/asyncio-subprocess.html#asyncio.asyncio.subprocess.Process.wait
|
||||
#
|
||||
# If it starts hanging, will need to switch s/wait/communicate/ - so perhaps for debug we will enable
|
||||
# `wait` as it's easier to see in real time, but for normal runs use `communicate`
|
||||
return _RunOutput(await p.wait(), out, err)
|
||||
|
||||
|
||||
def execute_async_std(cmd, env=None, stdin=None, timeout=None, quiet=False, echo=False) -> _RunOutput:
|
||||
loop = asyncio.get_event_loop()
|
||||
result = loop.run_until_complete(
|
||||
_stream_subprocess(cmd, env=env, stdin=stdin, timeout=timeout, quiet=quiet, echo=echo)
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
@@ -67,10 +67,10 @@ class ExamplesTests(TestCasePlus):
|
||||
testargs = f"""
|
||||
run_glue.py
|
||||
--model_name_or_path distilbert-base-uncased
|
||||
--data_dir ./tests/fixtures/tests_samples/MRPC/
|
||||
--output_dir {tmp_dir}
|
||||
--overwrite_output_dir
|
||||
--task_name mrpc
|
||||
--train_file ./tests/fixtures/tests_samples/MRPC/train.csv
|
||||
--validation_file ./tests/fixtures/tests_samples/MRPC/dev.csv
|
||||
--do_train
|
||||
--do_eval
|
||||
--per_device_train_batch_size=2
|
||||
|
||||
@@ -44,8 +44,7 @@ class TorchXLAExamplesTests(unittest.TestCase):
|
||||
transformers/examples/text-classification/run_glue.py
|
||||
--do_train
|
||||
--do_eval
|
||||
--task_name=MRPC
|
||||
--data_dir=/datasets/glue_data/MRPC
|
||||
--task_name=mrpc
|
||||
--cache_dir=./cache_dir
|
||||
--num_train_epochs=1
|
||||
--max_seq_length=128
|
||||
|
||||
@@ -74,18 +74,10 @@ between different runs. We report the median on 5 runs (with different seeds) fo
|
||||
| WNLI | Accuracy | 45.07 |
|
||||
|
||||
Some of these results are significantly different from the ones reported on the test set
|
||||
of GLUE benchmark on the website. For QQP and WNLI, please refer to [FAQ #12](https://gluebenchmark.com/faq) on the webite.
|
||||
|
||||
Before running any one of these GLUE tasks you should download the
|
||||
[GLUE data](https://gluebenchmark.com/tasks) by running the following lines at the root of the repo
|
||||
```
|
||||
python utils/download_glue_data.py --data_dir /path/to/glue --tasks all
|
||||
```
|
||||
|
||||
after replacing *path/to/glue* with a value that you like. Then you can run
|
||||
of GLUE benchmark on the website. For QQP and WNLI, please refer to [FAQ #12](https://gluebenchmark.com/faq) on the
|
||||
website.
|
||||
|
||||
```bash
|
||||
export GLUE_DIR=/path/to/glue
|
||||
export TASK_NAME=MRPC
|
||||
|
||||
python run_glue.py \
|
||||
@@ -93,7 +85,6 @@ python run_glue.py \
|
||||
--task_name $TASK_NAME \
|
||||
--do_train \
|
||||
--do_eval \
|
||||
--data_dir $GLUE_DIR/$TASK_NAME \
|
||||
--max_seq_length 128 \
|
||||
--per_device_train_batch_size 32 \
|
||||
--learning_rate 2e-5 \
|
||||
@@ -114,69 +105,33 @@ since the data processor for each task inherits from the base class DataProcesso
|
||||
|
||||
## Running on TPUs in PyTorch
|
||||
|
||||
**Update**: read the more up-to-date [Running on TPUs](../README.md#running-on-tpus) in the main README.md instead.
|
||||
|
||||
Even when running PyTorch, you can accelerate your workloads on Google's TPUs, using `pytorch/xla`. For information on how to setup your TPU environment refer to the
|
||||
Even when running PyTorch, you can accelerate your workloads on Google's TPUs, using `pytorch/xla`. For information on
|
||||
how to setup your TPU environment refer to the
|
||||
[pytorch/xla README](https://github.com/pytorch/xla/blob/master/README.md).
|
||||
|
||||
The following are some examples of running the `*_tpu.py` finetuning scripts on TPUs. All steps for data preparation are
|
||||
identical to your normal GPU + Huggingface setup.
|
||||
|
||||
For running your GLUE task on MNLI dataset you can run something like the following:
|
||||
For running your GLUE task on MNLI dataset you can run something like the following form the root of the transformers
|
||||
repo:
|
||||
|
||||
```
|
||||
export XRT_TPU_CONFIG="tpu_worker;0;$TPU_IP_ADDRESS:8470"
|
||||
export GLUE_DIR=/path/to/glue
|
||||
export TASK_NAME=MNLI
|
||||
|
||||
python run_glue_tpu.py \
|
||||
--model_name_or_path bert-base-cased \
|
||||
--task_name $TASK_NAME \
|
||||
python examples/xla_spawn.py \
|
||||
--num_cores=8 \
|
||||
transformers/examples/text-classification/run_glue.py \
|
||||
--do_train \
|
||||
--do_eval \
|
||||
--data_dir $GLUE_DIR/$TASK_NAME \
|
||||
--max_seq_length 128 \
|
||||
--train_batch_size 32 \
|
||||
--learning_rate 3e-5 \
|
||||
--num_train_epochs 3.0 \
|
||||
--output_dir /tmp/$TASK_NAME \
|
||||
--task_name=mrpc \
|
||||
--num_train_epochs=3 \
|
||||
--max_seq_length=128 \
|
||||
--learning_rate=5e-5 \
|
||||
--output_dir=/tmp/mrpc \
|
||||
--overwrite_output_dir \
|
||||
--logging_steps 50 \
|
||||
--save_steps 200 \
|
||||
--num_cores=8
|
||||
--logging_steps=5 \
|
||||
--save_steps=5 \
|
||||
--tpu_metrics_debug \
|
||||
--model_name_or_path=bert-base-cased \
|
||||
--per_device_train_batch_size=64 \
|
||||
--per_device_eval_batch_size=64
|
||||
```
|
||||
|
||||
### MRPC
|
||||
|
||||
#### Fine-tuning example
|
||||
|
||||
The following examples fine-tune BERT on the Microsoft Research Paraphrase Corpus (MRPC) corpus and runs in less
|
||||
than 10 minutes on a single K-80 and in 27 seconds (!) on single tesla V100 16GB with apex installed.
|
||||
|
||||
Before running any one of these GLUE tasks you should download the
|
||||
[GLUE data](https://gluebenchmark.com/tasks) by running
|
||||
[this script](https://gist.github.com/W4ngatang/60c2bdb54d156a41194446737ce03e2e)
|
||||
and unpack it to some directory `$GLUE_DIR`.
|
||||
|
||||
```bash
|
||||
export GLUE_DIR=/path/to/glue
|
||||
|
||||
python run_glue.py \
|
||||
--model_name_or_path bert-base-cased \
|
||||
--task_name MRPC \
|
||||
--do_train \
|
||||
--do_eval \
|
||||
--data_dir $GLUE_DIR/MRPC/ \
|
||||
--max_seq_length 128 \
|
||||
--per_device_train_batch_size 32 \
|
||||
--learning_rate 2e-5 \
|
||||
--num_train_epochs 3.0 \
|
||||
--output_dir /tmp/mrpc_output/
|
||||
```
|
||||
|
||||
Our test ran on a few seeds with [the original implementation hyper-
|
||||
parameters](https://github.com/google-research/bert#sentence-and-sentence-pair-classification-tasks) gave evaluation
|
||||
results between 84% and 88%.
|
||||
|
||||
#### Using Apex and mixed-precision
|
||||
|
||||
@@ -184,14 +139,12 @@ Using Apex and 16 bit precision, the fine-tuning on MRPC only takes 27 seconds.
|
||||
[apex](https://github.com/NVIDIA/apex), then run the following example:
|
||||
|
||||
```bash
|
||||
export GLUE_DIR=/path/to/glue
|
||||
|
||||
python run_glue.py \
|
||||
--model_name_or_path bert-base-cased \
|
||||
--task_name MRPC \
|
||||
--do_train \
|
||||
--do_eval \
|
||||
--data_dir $GLUE_DIR/MRPC/ \
|
||||
--max_seq_length 128 \
|
||||
--per_device_train_batch_size 32 \
|
||||
--learning_rate 2e-5 \
|
||||
@@ -206,15 +159,13 @@ Here is an example using distributed training on 8 V100 GPUs. The model used is
|
||||
reaches F1 > 92 on MRPC.
|
||||
|
||||
```bash
|
||||
export GLUE_DIR=/path/to/glue
|
||||
|
||||
python -m torch.distributed.launch \
|
||||
--nproc_per_node 8 run_glue.py \
|
||||
--model_name_or_path bert-base-cased \
|
||||
--task_name MRPC \
|
||||
--task_name mrpc \
|
||||
--do_train \
|
||||
--do_eval \
|
||||
--data_dir $GLUE_DIR/MRPC/ \
|
||||
--max_seq_length 128 \
|
||||
--per_device_train_batch_size 8 \
|
||||
--learning_rate 2e-5 \
|
||||
@@ -246,7 +197,6 @@ python -m torch.distributed.launch \
|
||||
--task_name mnli \
|
||||
--do_train \
|
||||
--do_eval \
|
||||
--data_dir $GLUE_DIR/MNLI/ \
|
||||
--max_seq_length 128 \
|
||||
--per_device_train_batch_size 8 \
|
||||
--learning_rate 2e-5 \
|
||||
@@ -272,7 +222,9 @@ The results are the following:
|
||||
|
||||
# Run PyTorch version using PyTorch-Lightning
|
||||
|
||||
Run `bash run_pl.sh` from the `glue` directory. This will also install `pytorch-lightning` and the requirements in `examples/requirements.txt`. It is a shell pipeline that will automatically download, pre-process the data and run the specified models. Logs are saved in `lightning_logs` directory.
|
||||
Run `bash run_pl.sh` from the `glue` directory. This will also install `pytorch-lightning` and the requirements in
|
||||
`examples/requirements.txt`. It is a shell pipeline that will automatically download, preprocess the data and run the
|
||||
specified models. Logs are saved in `lightning_logs` directory.
|
||||
|
||||
Pass `--gpus` flag to change the number of GPUs. Default uses 1. At the end, the expected results are:
|
||||
|
||||
|
||||
@@ -14,33 +14,101 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
""" Finetuning the library models for sequence classification on GLUE."""
|
||||
# You can also adapt this script on your own text classification task. Pointers for this are left as comments.
|
||||
|
||||
|
||||
import dataclasses
|
||||
import logging
|
||||
import os
|
||||
import random
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Callable, Dict, Optional
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
from datasets import load_dataset, load_metric
|
||||
|
||||
from transformers import AutoConfig, AutoModelForSequenceClassification, AutoTokenizer, EvalPrediction, GlueDataset
|
||||
from transformers import GlueDataTrainingArguments as DataTrainingArguments
|
||||
import transformers
|
||||
from transformers import (
|
||||
AutoConfig,
|
||||
AutoModelForSequenceClassification,
|
||||
AutoTokenizer,
|
||||
EvalPrediction,
|
||||
HfArgumentParser,
|
||||
PretrainedConfig,
|
||||
Trainer,
|
||||
TrainingArguments,
|
||||
glue_compute_metrics,
|
||||
glue_output_modes,
|
||||
glue_tasks_num_labels,
|
||||
default_data_collator,
|
||||
set_seed,
|
||||
)
|
||||
from transformers.trainer_utils import is_main_process
|
||||
|
||||
|
||||
task_to_keys = {
|
||||
"cola": ("sentence", None),
|
||||
"mnli": ("premise", "hypothesis"),
|
||||
"mrpc": ("sentence1", "sentence2"),
|
||||
"qnli": ("question", "sentence"),
|
||||
"qqp": ("question1", "question2"),
|
||||
"rte": ("sentence1", "sentence2"),
|
||||
"sst2": ("sentence", None),
|
||||
"stsb": ("sentence1", "sentence2"),
|
||||
"wnli": ("sentence1", "sentence2"),
|
||||
}
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class DataTrainingArguments:
|
||||
"""
|
||||
Arguments pertaining to what data we are going to input our model for training and eval.
|
||||
|
||||
Using `HfArgumentParser` we can turn this class
|
||||
into argparse arguments to be able to specify them on
|
||||
the command line.
|
||||
"""
|
||||
|
||||
task_name: Optional[str] = field(
|
||||
default=None,
|
||||
metadata={"help": "The name of the task to train on: " + ", ".join(task_to_keys.keys())},
|
||||
)
|
||||
max_seq_length: int = field(
|
||||
default=128,
|
||||
metadata={
|
||||
"help": "The maximum total input sequence length after tokenization. Sequences longer "
|
||||
"than this will be truncated, sequences shorter will be padded."
|
||||
},
|
||||
)
|
||||
overwrite_cache: bool = field(
|
||||
default=False, metadata={"help": "Overwrite the cached preprocessed datasets or not."}
|
||||
)
|
||||
pad_to_max_length: bool = field(
|
||||
default=True,
|
||||
metadata={
|
||||
"help": "Whether to pad all samples to `max_seq_length`. "
|
||||
"If False, will pad the samples dynamically when batching to the maximum length in the batch."
|
||||
},
|
||||
)
|
||||
train_file: Optional[str] = field(
|
||||
default=None, metadata={"help": "A csv or a json file containing the training data."}
|
||||
)
|
||||
validation_file: Optional[str] = field(
|
||||
default=None, metadata={"help": "A csv or a json file containing the validation data."}
|
||||
)
|
||||
|
||||
def __post_init__(self):
|
||||
if self.task_name is not None:
|
||||
self.task_name = self.task_name.lower()
|
||||
if self.task_name not in task_to_keys.keys():
|
||||
raise ValueError("Unknown task, you should pick one in " + ",".join(task_to_keys.keys()))
|
||||
elif self.train_file is None or self.validation_file is None:
|
||||
raise ValueError("Need either a GLUE task or a training/validation file.")
|
||||
else:
|
||||
extension = self.train_file.split(".")[-1]
|
||||
assert extension in ["csv", "json"], "`train_file` should be a csv or a json file."
|
||||
extension = self.validation_file.split(".")[-1]
|
||||
assert extension in ["csv", "json"], "`validation_file` should be a csv or a json file."
|
||||
|
||||
|
||||
@dataclass
|
||||
class ModelArguments:
|
||||
"""
|
||||
@@ -59,6 +127,10 @@ class ModelArguments:
|
||||
cache_dir: Optional[str] = field(
|
||||
default=None, metadata={"help": "Where do you want to store the pretrained models downloaded from s3"}
|
||||
)
|
||||
use_fast_tokenizer: bool = field(
|
||||
default=True,
|
||||
metadata={"help": "Whether to use one of the fast tokenizer (backed by the tokenizers library) or not."},
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
@@ -67,7 +139,6 @@ def main():
|
||||
# We now keep distinct sets of args, for a cleaner separation of concerns.
|
||||
|
||||
parser = HfArgumentParser((ModelArguments, DataTrainingArguments, TrainingArguments))
|
||||
|
||||
if len(sys.argv) == 2 and sys.argv[1].endswith(".json"):
|
||||
# If we pass only one argument to the script and it's the path to a json file,
|
||||
# let's parse it to get our arguments.
|
||||
@@ -82,40 +153,82 @@ def main():
|
||||
and not training_args.overwrite_output_dir
|
||||
):
|
||||
raise ValueError(
|
||||
f"Output directory ({training_args.output_dir}) already exists and is not empty. Use --overwrite_output_dir to overcome."
|
||||
f"Output directory ({training_args.output_dir}) already exists and is not empty. "
|
||||
"Use --overwrite_output_dir to overcome."
|
||||
)
|
||||
|
||||
# Setup logging
|
||||
logging.basicConfig(
|
||||
format="%(asctime)s - %(levelname)s - %(name)s - %(message)s",
|
||||
datefmt="%m/%d/%Y %H:%M:%S",
|
||||
level=logging.INFO if training_args.local_rank in [-1, 0] else logging.WARN,
|
||||
level=logging.INFO if is_main_process(training_args.local_rank) else logging.WARN,
|
||||
)
|
||||
logger.warning(
|
||||
"Process rank: %s, device: %s, n_gpu: %s, distributed training: %s, 16-bits training: %s",
|
||||
training_args.local_rank,
|
||||
training_args.device,
|
||||
training_args.n_gpu,
|
||||
bool(training_args.local_rank != -1),
|
||||
training_args.fp16,
|
||||
)
|
||||
logger.info("Training/evaluation parameters %s", training_args)
|
||||
|
||||
# Set seed
|
||||
# Log on each process the small summary:
|
||||
logger.warning(
|
||||
f"Process rank: {training_args.local_rank}, device: {training_args.device}, n_gpu: {training_args.n_gpu}"
|
||||
+ f"distributed training: {bool(training_args.local_rank != -1)}, 16-bits training: {training_args.fp16}"
|
||||
)
|
||||
# Set the verbosity to info of the Transformers logger (on main process only):
|
||||
if is_main_process(training_args.local_rank):
|
||||
transformers.utils.logging.set_verbosity_info()
|
||||
logger.info(f"Training/evaluation parameters {training_args}")
|
||||
|
||||
# Set seed before initializing model.
|
||||
set_seed(training_args.seed)
|
||||
|
||||
try:
|
||||
num_labels = glue_tasks_num_labels[data_args.task_name]
|
||||
output_mode = glue_output_modes[data_args.task_name]
|
||||
except KeyError:
|
||||
raise ValueError("Task not found: %s" % (data_args.task_name))
|
||||
# Get the datasets: you can either provide your own CSV/JSON training and evaluation files (see below)
|
||||
# or specify a GLUE benchmark task (the dataset will be downloaded automatically from the datasets Hub
|
||||
#
|
||||
# For CSV/JSON files, this script will use as labels the column called 'label' and as pair of sentences the
|
||||
# sentences in columns called 'sentence1' and 'sentence2' if such column exists or the first two columns not named
|
||||
# label if at least two columns are provided.
|
||||
#
|
||||
# If the CSVs/JSONs contain only one non-label column, the script does single sentence classification on this
|
||||
# single column. You can easily tweak this behavior (see below)
|
||||
#
|
||||
# In distributed training, the load_dataset function guarantee that only one local process can concurrently
|
||||
# download the dataset.
|
||||
if data_args.task_name is not None:
|
||||
# Downloading and loading a dataset from the hub.
|
||||
datasets = load_dataset("glue", data_args.task_name)
|
||||
elif data_args.train_file.endswith(".csv"):
|
||||
# Loading a dataset from local csv files
|
||||
datasets = load_dataset(
|
||||
"csv", data_files={"train": data_args.train_file, "validation": data_args.validation_file}
|
||||
)
|
||||
else:
|
||||
# Loading a dataset from local json files
|
||||
datasets = load_dataset(
|
||||
"json", data_files={"train": data_args.train_file, "validation": data_args.validation_file}
|
||||
)
|
||||
# See more about loading any type of standard or custom dataset at
|
||||
# https://huggingface.co/docs/datasets/loading_datasets.html.
|
||||
|
||||
# Labels
|
||||
if data_args.task_name is not None:
|
||||
is_regression = data_args.task_name == "stsb"
|
||||
if not is_regression:
|
||||
label_list = datasets["train"].features["label"].names
|
||||
num_labels = len(label_list)
|
||||
else:
|
||||
num_labels = 1
|
||||
else:
|
||||
# Trying to have good defaults here, don't hesitate to tweak to your needs.
|
||||
is_regression = datasets["train"].features["label"].dtype in ["float32", "float64"]
|
||||
if is_regression:
|
||||
num_labels = 1
|
||||
else:
|
||||
# A useful fast method:
|
||||
# https://huggingface.co/docs/datasets/package_reference/main_classes.html#datasets.Dataset.unique
|
||||
label_list = datasets["train"].unique("label")
|
||||
label_list.sort() # Let's sort it for determinism
|
||||
num_labels = len(label_list)
|
||||
|
||||
# Load pretrained model and tokenizer
|
||||
#
|
||||
# Distributed training:
|
||||
# The .from_pretrained methods guarantee that only one local process can concurrently
|
||||
# In distributed training, the .from_pretrained methods guarantee that only one local process can concurrently
|
||||
# download model & vocab.
|
||||
|
||||
config = AutoConfig.from_pretrained(
|
||||
model_args.config_name if model_args.config_name else model_args.model_name_or_path,
|
||||
num_labels=num_labels,
|
||||
@@ -125,6 +238,7 @@ def main():
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
model_args.tokenizer_name if model_args.tokenizer_name else model_args.model_name_or_path,
|
||||
cache_dir=model_args.cache_dir,
|
||||
use_fast=model_args.use_fast_tokenizer,
|
||||
)
|
||||
model = AutoModelForSequenceClassification.from_pretrained(
|
||||
model_args.model_name_or_path,
|
||||
@@ -133,39 +247,103 @@ def main():
|
||||
cache_dir=model_args.cache_dir,
|
||||
)
|
||||
|
||||
# Get datasets
|
||||
train_dataset = (
|
||||
GlueDataset(data_args, tokenizer=tokenizer, cache_dir=model_args.cache_dir) if training_args.do_train else None
|
||||
)
|
||||
eval_dataset = (
|
||||
GlueDataset(data_args, tokenizer=tokenizer, mode="dev", cache_dir=model_args.cache_dir)
|
||||
if training_args.do_eval
|
||||
else None
|
||||
)
|
||||
test_dataset = (
|
||||
GlueDataset(data_args, tokenizer=tokenizer, mode="test", cache_dir=model_args.cache_dir)
|
||||
if training_args.do_predict
|
||||
else None
|
||||
)
|
||||
# Preprocessing the datasets
|
||||
if data_args.task_name is not None:
|
||||
sentence1_key, sentence2_key = task_to_keys[data_args.task_name]
|
||||
else:
|
||||
# Again, we try to have some nice defaults but don't hesitate to tweak to your use case.
|
||||
non_label_column_names = [name for name in datasets["train"].column_names if name != "label"]
|
||||
if "sentence1" in non_label_column_names and "sentence2" in non_label_column_names:
|
||||
sentence1_key, sentence2_key = "sentence1", "sentence2"
|
||||
else:
|
||||
if len(non_label_column_names) >= 2:
|
||||
sentence1_key, sentence2_key = non_label_column_names[:2]
|
||||
else:
|
||||
sentence1_key, sentence2_key = non_label_column_names[0], None
|
||||
|
||||
def build_compute_metrics_fn(task_name: str) -> Callable[[EvalPrediction], Dict]:
|
||||
def compute_metrics_fn(p: EvalPrediction):
|
||||
preds = p.predictions[0] if isinstance(p.predictions, tuple) else p.predictions
|
||||
if output_mode == "classification":
|
||||
preds = np.argmax(preds, axis=1)
|
||||
else: # regression
|
||||
preds = np.squeeze(preds)
|
||||
return glue_compute_metrics(task_name, preds, p.label_ids)
|
||||
# Padding strategy
|
||||
if data_args.pad_to_max_length:
|
||||
padding = "max_length"
|
||||
max_length = data_args.max_seq_length
|
||||
else:
|
||||
# We will pad later, dynamically at batch creation, to the max sequence length in each batch
|
||||
padding = False
|
||||
max_length = None
|
||||
|
||||
return compute_metrics_fn
|
||||
# Some models have set the order of the labels to use, so let's make sure we do use it.
|
||||
label_to_id = None
|
||||
if (
|
||||
model.config.label2id != PretrainedConfig(num_labels=num_labels).label2id
|
||||
and data_args.task_name is not None
|
||||
and is_regression
|
||||
):
|
||||
# Some have all caps in their config, some don't.
|
||||
label_name_to_id = {k.lower(): v for k, v in model.config.label2id.items()}
|
||||
if list(sorted(label_name_to_id.keys())) == list(sorted(label_list)):
|
||||
label_to_id = {i: label_name_to_id[label_list[i]] for i in range(num_labels)}
|
||||
else:
|
||||
logger.warn(
|
||||
"Your model seems to have been trained with labels, but they don't match the dataset: ",
|
||||
f"model labels: {list(sorted(label_name_to_id.keys()))}, dataset labels: {list(sorted(label_list))}."
|
||||
"\nIgnoring the model labels as a result.",
|
||||
)
|
||||
elif data_args.task_name is None:
|
||||
label_to_id = {v: i for i, v in enumerate(label_list)}
|
||||
|
||||
def preprocess_function(examples):
|
||||
# Tokenize the texts
|
||||
args = (
|
||||
(examples[sentence1_key],) if sentence2_key is None else (examples[sentence1_key], examples[sentence2_key])
|
||||
)
|
||||
result = tokenizer(*args, padding=padding, max_length=max_length, truncation=True)
|
||||
|
||||
# Map labels to IDs (not necessary for GLUE tasks)
|
||||
if label_to_id is not None and "label" in examples:
|
||||
result["label"] = [label_to_id[l] for l in examples["label"]]
|
||||
return result
|
||||
|
||||
datasets = datasets.map(preprocess_function, batched=True, load_from_cache_file=not data_args.overwrite_cache)
|
||||
|
||||
train_dataset = datasets["train"]
|
||||
eval_dataset = datasets["validation_matched" if data_args.task_name == "mnli" else "validation"]
|
||||
if data_args.task_name is not None:
|
||||
test_dataset = datasets["test_matched" if data_args.task_name == "mnli" else "test"]
|
||||
|
||||
# Log a few random samples from the training set:
|
||||
for index in random.sample(range(len(train_dataset)), 3):
|
||||
logger.info(f"Sample {index} of the training set: {train_dataset[index]}.")
|
||||
|
||||
# Get the metric function
|
||||
if data_args.task_name is not None:
|
||||
metric = load_metric("glue", data_args.task_name)
|
||||
# TODO: When datasets metrics include regular accuracy, make an else here and remove special branch from
|
||||
# compute_metrics
|
||||
|
||||
# You can define your custom compute_metrics function. It takes an `EvalPrediction` object (a namedtuple with a
|
||||
# predictions and label_ids field) and has to return a dictionary string to float.
|
||||
def compute_metrics(p: EvalPrediction):
|
||||
preds = p.predictions[0] if isinstance(p.predictions, tuple) else p.predictions
|
||||
preds = np.squeeze(preds) if is_regression else np.argmax(preds, axis=1)
|
||||
if data_args.task_name is not None:
|
||||
result = metric.compute(predictions=preds, references=p.label_ids)
|
||||
if len(result) > 1:
|
||||
result["combined_score"] = np.mean(list(result.values())).item()
|
||||
return result
|
||||
elif is_regression:
|
||||
return {"mse": ((preds - p.label_ids) ** 2).mean().item()}
|
||||
else:
|
||||
return {"accuracy": (preds == p.label_ids).astype(np.float32).mean().item()}
|
||||
|
||||
# Initialize our Trainer
|
||||
trainer = Trainer(
|
||||
model=model,
|
||||
args=training_args,
|
||||
train_dataset=train_dataset,
|
||||
eval_dataset=eval_dataset,
|
||||
compute_metrics=build_compute_metrics_fn(data_args.task_name),
|
||||
eval_dataset=eval_dataset if training_args.do_eval else None,
|
||||
compute_metrics=compute_metrics,
|
||||
tokenizer=tokenizer,
|
||||
# Data collator will default to DataCollatorWithPadding, so we change it if we already did the padding.
|
||||
data_collator=default_data_collator if data_args.pad_to_max_length else None,
|
||||
)
|
||||
|
||||
# Training
|
||||
@@ -173,11 +351,7 @@ def main():
|
||||
trainer.train(
|
||||
model_path=model_args.model_name_or_path if os.path.isdir(model_args.model_name_or_path) else None
|
||||
)
|
||||
trainer.save_model()
|
||||
# For convenience, we also re-save the tokenizer to the same directory,
|
||||
# so that you can share your model easily on huggingface.co/models =)
|
||||
if trainer.is_world_master():
|
||||
tokenizer.save_pretrained(training_args.output_dir)
|
||||
trainer.save_model() # Saves the tokenizer too for easy upload
|
||||
|
||||
# Evaluation
|
||||
eval_results = {}
|
||||
@@ -185,56 +359,52 @@ def main():
|
||||
logger.info("*** Evaluate ***")
|
||||
|
||||
# Loop to handle MNLI double evaluation (matched, mis-matched)
|
||||
tasks = [data_args.task_name]
|
||||
eval_datasets = [eval_dataset]
|
||||
if data_args.task_name == "mnli":
|
||||
mnli_mm_data_args = dataclasses.replace(data_args, task_name="mnli-mm")
|
||||
eval_datasets.append(
|
||||
GlueDataset(mnli_mm_data_args, tokenizer=tokenizer, mode="dev", cache_dir=model_args.cache_dir)
|
||||
)
|
||||
tasks.append("mnli-mm")
|
||||
eval_datasets.append(datasets["validation_mismatched"])
|
||||
|
||||
for eval_dataset in eval_datasets:
|
||||
trainer.compute_metrics = build_compute_metrics_fn(eval_dataset.args.task_name)
|
||||
for eval_dataset, task in zip(eval_datasets, tasks):
|
||||
eval_result = trainer.evaluate(eval_dataset=eval_dataset)
|
||||
|
||||
output_eval_file = os.path.join(
|
||||
training_args.output_dir, f"eval_results_{eval_dataset.args.task_name}.txt"
|
||||
)
|
||||
if trainer.is_world_master():
|
||||
output_eval_file = os.path.join(training_args.output_dir, f"eval_results_{task}.txt")
|
||||
if trainer.is_world_process_zero():
|
||||
with open(output_eval_file, "w") as writer:
|
||||
logger.info("***** Eval results {} *****".format(eval_dataset.args.task_name))
|
||||
logger.info(f"***** Eval results {task} *****")
|
||||
for key, value in eval_result.items():
|
||||
logger.info(" %s = %s", key, value)
|
||||
writer.write("%s = %s\n" % (key, value))
|
||||
logger.info(f" {key} = {value}")
|
||||
writer.write(f"{key} = {value}\n")
|
||||
|
||||
eval_results.update(eval_result)
|
||||
|
||||
if training_args.do_predict:
|
||||
logging.info("*** Test ***")
|
||||
logger.info("*** Test ***")
|
||||
|
||||
# Loop to handle MNLI double evaluation (matched, mis-matched)
|
||||
tasks = [data_args.task_name]
|
||||
test_datasets = [test_dataset]
|
||||
if data_args.task_name == "mnli":
|
||||
mnli_mm_data_args = dataclasses.replace(data_args, task_name="mnli-mm")
|
||||
test_datasets.append(
|
||||
GlueDataset(mnli_mm_data_args, tokenizer=tokenizer, mode="test", cache_dir=model_args.cache_dir)
|
||||
)
|
||||
tasks.append("mnli-mm")
|
||||
test_datasets.append(datasets["test_mismatched"])
|
||||
|
||||
for test_dataset in test_datasets:
|
||||
for test_dataset, task in zip(test_datasets, tasks):
|
||||
# Removing the `label` columns because it contains -1 and Trainer won't like that.
|
||||
test_dataset.remove_columns_("label")
|
||||
predictions = trainer.predict(test_dataset=test_dataset).predictions
|
||||
if output_mode == "classification":
|
||||
predictions = np.argmax(predictions, axis=1)
|
||||
predictions = np.squeeze(predictions) if is_regression else np.argmax(predictions, axis=1)
|
||||
|
||||
output_test_file = os.path.join(
|
||||
training_args.output_dir, f"test_results_{test_dataset.args.task_name}.txt"
|
||||
)
|
||||
if trainer.is_world_master():
|
||||
output_test_file = os.path.join(training_args.output_dir, f"test_results_{task}.txt")
|
||||
if trainer.is_world_process_zero():
|
||||
with open(output_test_file, "w") as writer:
|
||||
logger.info("***** Test results {} *****".format(test_dataset.args.task_name))
|
||||
logger.info(f"***** Test results {task} *****")
|
||||
writer.write("index\tprediction\n")
|
||||
for index, item in enumerate(predictions):
|
||||
if output_mode == "regression":
|
||||
writer.write("%d\t%3.3f\n" % (index, item))
|
||||
if is_regression:
|
||||
writer.write(f"{index}\t{item:3.3f}\n")
|
||||
else:
|
||||
item = test_dataset.get_labels()[item]
|
||||
writer.write("%d\t%s\n" % (index, item))
|
||||
item = label_list[item]
|
||||
writer.write(f"{index}\t{item}\n")
|
||||
return eval_results
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
---
|
||||
language: da
|
||||
tags:
|
||||
- bert
|
||||
- masked-lm
|
||||
- lm-head
|
||||
license: cc-by-4.0
|
||||
datasets:
|
||||
- common_crawl
|
||||
- wikipedia
|
||||
pipeline_tag: fill-mask
|
||||
widget:
|
||||
- text: "København er [MASK] i Danmark."
|
||||
---
|
||||
|
||||
# Danish BERT (uncased) model
|
||||
|
||||
[BotXO.ai](https://www.botxo.ai/) developed this model. For data and training details see their [GitHub repository](https://github.com/botxo/nordic_bert).
|
||||
|
||||
The original model was trained in TensorFlow then I converted it to Pytorch using [transformers-cli](https://huggingface.co/transformers/converting_tensorflow_models.html?highlight=cli).
|
||||
|
||||
For TensorFlow version download here: https://www.dropbox.com/s/19cjaoqvv2jicq9/danish_bert_uncased_v2.zip?dl=1
|
||||
|
||||
|
||||
## Architecture
|
||||
|
||||
```python
|
||||
from transformers import AutoModelForPreTraining
|
||||
|
||||
model = AutoModelForPreTraining.from_pretrained("DJSammy/bert-base-danish-uncased_BotXO,ai")
|
||||
|
||||
params = list(model.named_parameters())
|
||||
print('danish_bert_uncased_v2 has {:} different named parameters.\n'.format(len(params)))
|
||||
|
||||
print('==== Embedding Layer ====\n')
|
||||
for p in params[0:5]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== First Transformer ====\n')
|
||||
for p in params[5:21]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== Last Transformer ====\n')
|
||||
for p in params[181:197]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== Output Layer ====\n')
|
||||
for p in params[197:]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
# danish_bert_uncased_v2 has 206 different named parameters.
|
||||
|
||||
# ==== Embedding Layer ====
|
||||
|
||||
# bert.embeddings.word_embeddings.weight (32000, 768)
|
||||
# bert.embeddings.position_embeddings.weight (512, 768)
|
||||
# bert.embeddings.token_type_embeddings.weight (2, 768)
|
||||
# bert.embeddings.LayerNorm.weight (768,)
|
||||
# bert.embeddings.LayerNorm.bias (768,)
|
||||
|
||||
# ==== First Transformer ====
|
||||
|
||||
# bert.encoder.layer.0.attention.self.query.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.query.bias (768,)
|
||||
# bert.encoder.layer.0.attention.self.key.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.key.bias (768,)
|
||||
# bert.encoder.layer.0.attention.self.value.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.value.bias (768,)
|
||||
# bert.encoder.layer.0.attention.output.dense.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.output.dense.bias (768,)
|
||||
# bert.encoder.layer.0.attention.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.0.attention.output.LayerNorm.bias (768,)
|
||||
# bert.encoder.layer.0.intermediate.dense.weight (3072, 768)
|
||||
# bert.encoder.layer.0.intermediate.dense.bias (3072,)
|
||||
# bert.encoder.layer.0.output.dense.weight (768, 3072)
|
||||
# bert.encoder.layer.0.output.dense.bias (768,)
|
||||
# bert.encoder.layer.0.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.0.output.LayerNorm.bias (768,)
|
||||
|
||||
# ==== Last Transformer ====
|
||||
|
||||
# bert.encoder.layer.11.attention.self.query.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.query.bias (768,)
|
||||
# bert.encoder.layer.11.attention.self.key.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.key.bias (768,)
|
||||
# bert.encoder.layer.11.attention.self.value.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.value.bias (768,)
|
||||
# bert.encoder.layer.11.attention.output.dense.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.output.dense.bias (768,)
|
||||
# bert.encoder.layer.11.attention.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.11.attention.output.LayerNorm.bias (768,)
|
||||
# bert.encoder.layer.11.intermediate.dense.weight (3072, 768)
|
||||
# bert.encoder.layer.11.intermediate.dense.bias (3072,)
|
||||
# bert.encoder.layer.11.output.dense.weight (768, 3072)
|
||||
# bert.encoder.layer.11.output.dense.bias (768,)
|
||||
# bert.encoder.layer.11.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.11.output.LayerNorm.bias (768,)
|
||||
|
||||
# ==== Output Layer ====
|
||||
|
||||
# bert.pooler.dense.weight (768, 768)
|
||||
# bert.pooler.dense.bias (768,)
|
||||
# cls.predictions.bias (32000,)
|
||||
# cls.predictions.transform.dense.weight (768, 768)
|
||||
# cls.predictions.transform.dense.bias (768,)
|
||||
# cls.predictions.transform.LayerNorm.weight (768,)
|
||||
# cls.predictions.transform.LayerNorm.bias (768,)
|
||||
# cls.seq_relationship.weight (2, 768)
|
||||
# cls.seq_relationship.bias (2,)
|
||||
```
|
||||
|
||||
## Example Pipeline
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
unmasker = pipeline('fill-mask', model='DJSammy/bert-base-danish-uncased_BotXO,ai')
|
||||
|
||||
unmasker('København er [MASK] i Danmark.')
|
||||
|
||||
# Copenhagen is the [MASK] of Denmark.
|
||||
# =>
|
||||
|
||||
# [{'score': 0.788068950176239,
|
||||
# 'sequence': '[CLS] københavn er hovedstad i danmark. [SEP]',
|
||||
# 'token': 12610,
|
||||
# 'token_str': 'hovedstad'},
|
||||
# {'score': 0.07606703042984009,
|
||||
# 'sequence': '[CLS] københavn er hovedstaden i danmark. [SEP]',
|
||||
# 'token': 8108,
|
||||
# 'token_str': 'hovedstaden'},
|
||||
# {'score': 0.04299738258123398,
|
||||
# 'sequence': '[CLS] københavn er metropol i danmark. [SEP]',
|
||||
# 'token': 23305,
|
||||
# 'token_str': 'metropol'},
|
||||
# {'score': 0.008163209073245525,
|
||||
# 'sequence': '[CLS] københavn er ikke i danmark. [SEP]',
|
||||
# 'token': 89,
|
||||
# 'token_str': 'ikke'},
|
||||
# {'score': 0.006238455418497324,
|
||||
# 'sequence': '[CLS] københavn er ogsa i danmark. [SEP]',
|
||||
# 'token': 25253,
|
||||
# 'token_str': 'ogsa'}]
|
||||
```
|
||||
@@ -0,0 +1,47 @@
|
||||
## About the model
|
||||
|
||||
The model has been trained on a collection of 500k articles with headings. Its purpose is to create a one-line heading suitable for the given article.
|
||||
|
||||
Sample code with a WikiNews article:
|
||||
|
||||
```python
|
||||
import torch
|
||||
from transformers import T5ForConditionalGeneration,T5Tokenizer
|
||||
|
||||
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
||||
|
||||
model = T5ForConditionalGeneration.from_pretrained("Michau/t5-base-en-generate-headline")
|
||||
tokenizer = T5Tokenizer.from_pretrained("Michau/t5-base-en-generate-headline")
|
||||
model = model.to(device)
|
||||
|
||||
article = '''
|
||||
Very early yesterday morning, the United States President Donald Trump reported he and his wife First Lady Melania Trump tested positive for COVID-19. Officials said the Trumps' 14-year-old son Barron tested negative as did First Family and Senior Advisors Jared Kushner and Ivanka Trump.
|
||||
Trump took to social media, posting at 12:54 am local time (0454 UTC) on Twitter, "Tonight, [Melania] and I tested positive for COVID-19. We will begin our quarantine and recovery process immediately. We will get through this TOGETHER!" Yesterday afternoon Marine One landed on the White House's South Lawn flying Trump to Walter Reed National Military Medical Center (WRNMMC) in Bethesda, Maryland.
|
||||
Reports said both were showing "mild symptoms". Senior administration officials were tested as people were informed of the positive test. Senior advisor Hope Hicks had tested positive on Thursday.
|
||||
Presidential physician Sean Conley issued a statement saying Trump has been given zinc, vitamin D, Pepcid and a daily Aspirin. Conley also gave a single dose of the experimental polyclonal antibodies drug from Regeneron Pharmaceuticals.
|
||||
According to official statements, Trump, now operating from the WRNMMC, is to continue performing his duties as president during a 14-day quarantine. In the event of Trump becoming incapacitated, Vice President Mike Pence could take over the duties of president via the 25th Amendment of the US Constitution. The Pence family all tested negative as of yesterday and there were no changes regarding Pence's campaign events.
|
||||
'''
|
||||
|
||||
text = "headline: " + article
|
||||
|
||||
max_len = 256
|
||||
|
||||
encoding = tokenizer.encode_plus(text, return_tensors = "pt")
|
||||
input_ids = encoding["input_ids"].to(device)
|
||||
attention_masks = encoding["attention_mask"].to(device)
|
||||
|
||||
beam_outputs = model.generate(
|
||||
input_ids = input_ids,
|
||||
attention_mask = attention_masks,
|
||||
max_length = 64,
|
||||
num_beams = 3,
|
||||
early_stopping = True,
|
||||
)
|
||||
|
||||
result = tokenizer.decode(beam_outputs[0])
|
||||
print(result)
|
||||
```
|
||||
|
||||
Result:
|
||||
|
||||
```Trump and First Lady Melania Test Positive for COVID-19```
|
||||
@@ -4,44 +4,4 @@ license: mit
|
||||
---
|
||||
|
||||
# bert-german-dbmdz-uncased-sentence-stsb
|
||||
|
||||
## How to use
|
||||
**The usage description above - provided by Hugging Face - is wrong! Please use this:**
|
||||
|
||||
Install the `sentence-transformers` package. See here: <https://github.com/UKPLab/sentence-transformers>
|
||||
```python
|
||||
from sentence_transformers import models
|
||||
from sentence_transformers import SentenceTransformer
|
||||
|
||||
# load BERT model from Hugging Face
|
||||
word_embedding_model = models.Transformer(
|
||||
'T-Systems-onsite/bert-german-dbmdz-uncased-sentence-stsb')
|
||||
|
||||
# Apply mean pooling to get one fixed sized sentence vector
|
||||
pooling_model = models.Pooling(word_embedding_model.get_word_embedding_dimension(),
|
||||
pooling_mode_mean_tokens=True,
|
||||
pooling_mode_cls_token=False,
|
||||
pooling_mode_max_tokens=False)
|
||||
|
||||
# join BERT model and pooling to get the sentence transformer
|
||||
model = SentenceTransformer(modules=[word_embedding_model, pooling_model])
|
||||
```
|
||||
|
||||
## Model description
|
||||
This is a German [sentence embedding](https://github.com/UKPLab/sentence-transformers) trained on the [German STSbenchmark Dataset](https://github.com/t-systems-on-site-services-gmbh/german-STSbenchmark). It was trained from [Philip May](https://eniak.de/) and open-sourced by [T-Systems-onsite](https://www.t-systems-onsite.de/).The base language model is the [dbmdz/bert-base-german-uncased](https://huggingface.co/dbmdz/bert-base-german-uncased) from [Bayerische Staatsbibliothek ](https://huggingface.co/dbmdz).
|
||||
|
||||
## Intended uses
|
||||
> Sentence-BERT (SBERT) is a modification of the pretrained BERT network that use siamese and triplet network structures to derive semantically mean-ingful sentence embeddings that can be compared using cosine-similarity. This reduces the effort for finding the most similar pair from 65hours with BERT / RoBERTa to about 5 seconds with SBERT, while maintaining the accuracy from BERT.
|
||||
|
||||
Source: [Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks](https://arxiv.org/abs/1908.10084)
|
||||
|
||||
## Training procedure
|
||||
We did an automatic hyperprameter optimization with [Optuna](https://github.com/optuna/optuna) and found the following hyperprameters:
|
||||
- batch_size = 5
|
||||
- num_epochs = 11
|
||||
- lr = 2.637549780860126e-05
|
||||
- eps = 5.0696075038683e-06
|
||||
- weight_decay = 0.02817210102940054
|
||||
- warmup_steps = 27.342745941760147 % of total steps
|
||||
|
||||
The final model was trained on the combination of all three datasets: `sts_de_dev.csv`, `sts_de_test.csv` and `sts_de_train.csv`
|
||||
**This model is outdated! Please use this improved version: <https://huggingface.co/T-Systems-onsite/german-roberta-sentence-transformer-v2>**
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
---
|
||||
language: de
|
||||
license: mit
|
||||
---
|
||||
|
||||
# German RoBERTa for Sentence Embeddings V2
|
||||
This model is intended to [compute sentence (text embeddings)](https://www.sbert.net/docs/usage/computing_sentence_embeddings.html) for German text. These embeddings can then be compared with [cosine-similarity](https://en.wikipedia.org/wiki/Cosine_similarity) to find sentences with a similar semantic meaning. For example this can be useful for [semantic textual similarity](https://www.sbert.net/docs/usage/semantic_textual_similarity.html), [semantic search](https://www.sbert.net/docs/usage/semantic_search.html), or [paraphrase mining](https://www.sbert.net/docs/usage/paraphrase_mining.html). To do this you have to use the [Sentence Transformers Python framework](https://github.com/UKPLab/sentence-transformers).
|
||||
|
||||
> Sentence-BERT (SBERT) is a modification of the pretrained BERT network that use siamese and triplet network structures to derive semantically mean-ingful sentence embeddings that can be compared using cosine-similarity. This reduces the effort for finding the most similar pair from 65hours with BERT / RoBERTa to about 5 seconds with SBERT, while maintaining the accuracy from BERT.
|
||||
|
||||
Source: [Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks](https://arxiv.org/abs/1908.10084)
|
||||
|
||||
This model is fine-tuned from [Philip May](https://eniak.de/) and open-sourced by [T-Systems-onsite](https://www.t-systems-onsite.de/). Special thanks to [Nils Reimers](https://www.nils-reimers.de/) for your awesome open-source work, the Sentence Transformers, the models and all your help on GitHub.
|
||||
|
||||
## How to use
|
||||
**The usage description above - provided by Hugging Face - is wrong for sentence embeddings! Please use this:**
|
||||
|
||||
To use this model install the `sentence-transformers` package (see here: <https://github.com/UKPLab/sentence-transformers>).
|
||||
|
||||
```python
|
||||
from sentence_transformers import SentenceTransformer
|
||||
model = SentenceTransformer('T-Systems-onsite/german-roberta-sentence-transformer-v2')
|
||||
```
|
||||
|
||||
For details of usage and examples see here:
|
||||
- [Computing Sentence Embeddings](https://www.sbert.net/docs/usage/computing_sentence_embeddings.html)
|
||||
- [Semantic Textual Similarity](https://www.sbert.net/docs/usage/semantic_textual_similarity.html)
|
||||
- [Paraphrase Mining](https://www.sbert.net/docs/usage/paraphrase_mining.html)
|
||||
- [Semantic Search](https://www.sbert.net/docs/usage/semantic_search.html)
|
||||
- [Cross-Encoders](https://www.sbert.net/docs/usage/cross-encoder.html)
|
||||
- [Examples on GitHub](https://github.com/UKPLab/sentence-transformers/tree/master/examples/applications)
|
||||
|
||||
## Training
|
||||
The base model is [xlm-roberta-base](https://huggingface.co/xlm-roberta-base). This model has been further trained by [Nils Reimers](https://www.nils-reimers.de/) on a large scale paraphrase dataset for 50+ languages. [Nils Reimers](https://www.nils-reimers.de/) about this [on GitHub](https://github.com/UKPLab/sentence-transformers/issues/509#issuecomment-712243280):
|
||||
|
||||
>A paper is upcoming for the paraphrase models.
|
||||
>
|
||||
>These models were trained on various datasets with Millions of examples for paraphrases, mainly derived from Wikipedia edit logs, paraphrases mined from Wikipedia and SimpleWiki, paraphrases from news reports, AllNLI-entailment pairs with in-batch-negative loss etc.
|
||||
>
|
||||
>In internal tests, they perform much better than the NLI+STSb models as they have see more and broader type of training data. NLI+STSb has the issue that they are rather narrow in their domain and do not contain any domain specific words / sentences (like from chemistry, computer science, math etc.). The paraphrase models has seen plenty of sentences from various domains.
|
||||
>
|
||||
>More details with the setup, all the datasets, and a wider evaluation will follow soon.
|
||||
|
||||
The resulting model called `xlm-r-distilroberta-base-paraphrase-v1` has been released here: <https://github.com/UKPLab/sentence-transformers/releases/tag/v0.3.8>
|
||||
|
||||
Building on this cross language model we fine-tuned it for German language on the deepl.com dataset of our [German STSbenchmark dataset](https://github.com/t-systems-on-site-services-gmbh/german-STSbenchmark).
|
||||
|
||||
We did an automatic hyperprameter search for 102 trials with [Optuna](https://github.com/optuna/optuna). Using crossvalidation on the deepl.com test and dev dataset we found the following best hyperprameters:
|
||||
- batch_size = 15
|
||||
- num_epochs = 4
|
||||
- lr = 2.2995320905210864e-05
|
||||
- eps = 1.8979875906303792e-06
|
||||
- weight_decay = 0.003314045812507563
|
||||
- warmup_steps_proportion = 0.46141685205829014
|
||||
|
||||
The final model was trained with these hyperparameters on the combination of `sts_de_train.csv` and `sts_de_dev.csv`. The `sts_de_test.csv` was left for testing. The AWS dataset has not been used.
|
||||
|
||||
# Evaluation
|
||||
The evaluation has been done on the test set of our [German STSbenchmark dataset](https://github.com/t-systems-on-site-services-gmbh/german-STSbenchmark). The code is available on [Colab](https://colab.research.google.com/drive/1aCWOqDQx953kEnQ5k4Qn7uiixokocOHv?usp=sharing). As the metric for evaluation we use the Spearman’s rank correlation between the cosine-similarity of the sentence embeddings and STSbenchmark labels.
|
||||
|
||||
| Model Name | Spearman rank correlation |
|
||||
|--------------------------------------|-----------------------------------|
|
||||
| xlm-r-distilroberta-base-paraphrase-v1 | 0.8079 |
|
||||
| xlm-r-100langs-bert-base-nli-stsb-mean-tokens | 0.8194 |
|
||||
| xlm-r-bert-base-nli-stsb-mean-tokens | 0.8194 |
|
||||
| **T-Systems-onsite/german-roberta-sentence-transformer-v2** | **0.8529** |
|
||||
@@ -0,0 +1,56 @@
|
||||
---
|
||||
language: "c++"
|
||||
tags:
|
||||
- exbert
|
||||
- authorship-identification
|
||||
- fire2020
|
||||
- pan2020
|
||||
- ai-soco
|
||||
- classification
|
||||
license: "mit"
|
||||
datasets:
|
||||
- ai-soco
|
||||
metrics:
|
||||
- accuracy
|
||||
---
|
||||
|
||||
# ai-soco-c++-roberta-small-clas
|
||||
|
||||
## Model description
|
||||
|
||||
`ai-soco-c++-roberta-small` model fine-tuned on [AI-SOCO](https://sites.google.com/view/ai-soco-2020) task.
|
||||
|
||||
#### How to use
|
||||
|
||||
You can use the model directly after tokenizing the text using the provided tokenizer with the model files.
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
The model is limited to C++ programming language only.
|
||||
|
||||
## Training data
|
||||
|
||||
The model initialized from [`ai-soco-c++-roberta-small`](https://github.com/huggingface/transformers/blob/master/model_cards/aliosm/ai-soco-c++-roberta-small) model and trained using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset to do text classification.
|
||||
|
||||
## Training procedure
|
||||
|
||||
The model trained on Google Colab platform using V100 GPU for 10 epochs, 32 batch size, 512 max sequence length (sequences larger than 512 were truncated). Each continues 4 spaces were converted to a single tab character (`\t`) before tokenization.
|
||||
|
||||
## Eval results
|
||||
|
||||
The model achieved 93.19%/92.88% accuracy on AI-SOCO task and ranked in the 4th place.
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{ai-soco-2020-fire,
|
||||
title = "Overview of the {PAN@FIRE} 2020 Task on {Authorship Identification of SOurce COde (AI-SOCO)}",
|
||||
author = "Fadel, Ali and Musleh, Husam and Tuffaha, Ibraheem and Al-Ayyoub, Mahmoud and Jararweh, Yaser and Benkhelifa, Elhadj and Rosso, Paolo",
|
||||
booktitle = "Proceedings of The 12th meeting of the Forum for Information Retrieval Evaluation (FIRE 2020)",
|
||||
year = "2020"
|
||||
}
|
||||
```
|
||||
|
||||
<a href="https://huggingface.co/exbert/?model=aliosm/ai-soco-c++-roberta-small-clas">
|
||||
<img width="300px" src="https://hf-dinosaur.huggingface.co/exbert/button.png">
|
||||
</a>
|
||||
@@ -0,0 +1,55 @@
|
||||
---
|
||||
language: "c++"
|
||||
tags:
|
||||
- exbert
|
||||
- authorship-identification
|
||||
- fire2020
|
||||
- pan2020
|
||||
- ai-soco
|
||||
license: "mit"
|
||||
datasets:
|
||||
- ai-soco
|
||||
metrics:
|
||||
- perplexity
|
||||
---
|
||||
|
||||
# ai-soco-c++-roberta-small
|
||||
|
||||
## Model description
|
||||
|
||||
From scratch pre-trained RoBERTa model with 6 layers and 12 attention heads using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset which consists of C++ codes crawled from CodeForces website.
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
The model can be used to do code classification, authorship identification and other downstream tasks on C++ programming language.
|
||||
|
||||
#### How to use
|
||||
|
||||
You can use the model directly after tokenizing the text using the provided tokenizer with the model files.
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
The model is limited to C++ programming language only.
|
||||
|
||||
## Training data
|
||||
|
||||
The model initialized randomly and trained using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset which contains 100K C++ source codes.
|
||||
|
||||
## Training procedure
|
||||
|
||||
The model trained on Google Colab platform with 8 TPU cores for 200 epochs, 16\*8 batch size, 512 max sequence length and MLM objective. Other parameters were defaulted to the values mentioned in [`run_language_modelling.py`](https://github.com/huggingface/transformers/blob/master/examples/language-modeling/run_language_modeling.py) script. Each continues 4 spaces were converted to a single tab character (`\t`) before tokenization.
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{ai-soco-2020-fire,
|
||||
title = "Overview of the {PAN@FIRE} 2020 Task on {Authorship Identification of SOurce COde (AI-SOCO)}",
|
||||
author = "Fadel, Ali and Musleh, Husam and Tuffaha, Ibraheem and Al-Ayyoub, Mahmoud and Jararweh, Yaser and Benkhelifa, Elhadj and Rosso, Paolo",
|
||||
booktitle = "Proceedings of The 12th meeting of the Forum for Information Retrieval Evaluation (FIRE 2020)",
|
||||
year = "2020"
|
||||
}
|
||||
```
|
||||
|
||||
<a href="https://huggingface.co/exbert/?model=aliosm/ai-soco-c++-roberta-small">
|
||||
<img width="300px" src="https://hf-dinosaur.huggingface.co/exbert/button.png">
|
||||
</a>
|
||||
@@ -0,0 +1,56 @@
|
||||
---
|
||||
language: "c++"
|
||||
tags:
|
||||
- exbert
|
||||
- authorship-identification
|
||||
- fire2020
|
||||
- pan2020
|
||||
- ai-soco
|
||||
- classification
|
||||
license: "mit"
|
||||
datasets:
|
||||
- ai-soco
|
||||
metrics:
|
||||
- accuracy
|
||||
---
|
||||
|
||||
# ai-soco-c++-roberta-tiny-96-clas
|
||||
|
||||
## Model description
|
||||
|
||||
`ai-soco-c++-roberta-tiny-96` model fine-tuned on [AI-SOCO](https://sites.google.com/view/ai-soco-2020) task.
|
||||
|
||||
#### How to use
|
||||
|
||||
You can use the model directly after tokenizing the text using the provided tokenizer with the model files.
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
The model is limited to C++ programming language only.
|
||||
|
||||
## Training data
|
||||
|
||||
The model initialized from [`ai-soco-c++-roberta-tiny-96`](https://github.com/huggingface/transformers/blob/master/model_cards/aliosm/ai-soco-c++-roberta-tiny-96) model and trained using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset to do text classification.
|
||||
|
||||
## Training procedure
|
||||
|
||||
The model trained on Google Colab platform using V100 GPU for 10 epochs, 16 batch size, 512 max sequence length (sequences larger than 512 were truncated). Each continues 4 spaces were converted to a single tab character (`\t`) before tokenization.
|
||||
|
||||
## Eval results
|
||||
|
||||
The model achieved 91.12%/91.02% accuracy on AI-SOCO task and ranked in the 7th place.
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{ai-soco-2020-fire,
|
||||
title = "Overview of the {PAN@FIRE} 2020 Task on {Authorship Identification of SOurce COde (AI-SOCO)}",
|
||||
author = "Fadel, Ali and Musleh, Husam and Tuffaha, Ibraheem and Al-Ayyoub, Mahmoud and Jararweh, Yaser and Benkhelifa, Elhadj and Rosso, Paolo",
|
||||
booktitle = "Proceedings of The 12th meeting of the Forum for Information Retrieval Evaluation (FIRE 2020)",
|
||||
year = "2020"
|
||||
}
|
||||
```
|
||||
|
||||
<a href="https://huggingface.co/exbert/?model=aliosm/ai-soco-c++-roberta-tiny-96-clas">
|
||||
<img width="300px" src="https://hf-dinosaur.huggingface.co/exbert/button.png">
|
||||
</a>
|
||||
@@ -0,0 +1,55 @@
|
||||
---
|
||||
language: "c++"
|
||||
tags:
|
||||
- exbert
|
||||
- authorship-identification
|
||||
- fire2020
|
||||
- pan2020
|
||||
- ai-soco
|
||||
license: "mit"
|
||||
datasets:
|
||||
- ai-soco
|
||||
metrics:
|
||||
- perplexity
|
||||
---
|
||||
|
||||
# ai-soco-c++-roberta-tiny-96
|
||||
|
||||
## Model description
|
||||
|
||||
From scratch pre-trained RoBERTa model with 1 layers and 96 attention heads using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset which consists of C++ codes crawled from CodeForces website.
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
The model can be used to do code classification, authorship identification and other downstream tasks on C++ programming language.
|
||||
|
||||
#### How to use
|
||||
|
||||
You can use the model directly after tokenizing the text using the provided tokenizer with the model files.
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
The model is limited to C++ programming language only.
|
||||
|
||||
## Training data
|
||||
|
||||
The model initialized randomly and trained using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset which contains 100K C++ source codes.
|
||||
|
||||
## Training procedure
|
||||
|
||||
The model trained on Google Colab platform with 8 TPU cores for 200 epochs, 16\*8 batch size, 512 max sequence length and MLM objective. Other parameters were defaulted to the values mentioned in [`run_language_modelling.py`](https://github.com/huggingface/transformers/blob/master/examples/language-modeling/run_language_modeling.py) script. Each continues 4 spaces were converted to a single tab character (`\t`) before tokenization.
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{ai-soco-2020-fire,
|
||||
title = "Overview of the {PAN@FIRE} 2020 Task on {Authorship Identification of SOurce COde (AI-SOCO)}",
|
||||
author = "Fadel, Ali and Musleh, Husam and Tuffaha, Ibraheem and Al-Ayyoub, Mahmoud and Jararweh, Yaser and Benkhelifa, Elhadj and Rosso, Paolo",
|
||||
booktitle = "Proceedings of The 12th meeting of the Forum for Information Retrieval Evaluation (FIRE 2020)",
|
||||
year = "2020"
|
||||
}
|
||||
```
|
||||
|
||||
<a href="https://huggingface.co/exbert/?model=aliosm/ai-soco-c++-roberta-tiny-96">
|
||||
<img width="300px" src="https://hf-dinosaur.huggingface.co/exbert/button.png">
|
||||
</a>
|
||||
@@ -0,0 +1,56 @@
|
||||
---
|
||||
language: "c++"
|
||||
tags:
|
||||
- exbert
|
||||
- authorship-identification
|
||||
- fire2020
|
||||
- pan2020
|
||||
- ai-soco
|
||||
- classification
|
||||
license: "mit"
|
||||
datasets:
|
||||
- ai-soco
|
||||
metrics:
|
||||
- accuracy
|
||||
---
|
||||
|
||||
# ai-soco-c++-roberta-tiny-clas
|
||||
|
||||
## Model description
|
||||
|
||||
`ai-soco-c++-roberta-tiny` model fine-tuned on [AI-SOCO](https://sites.google.com/view/ai-soco-2020) task.
|
||||
|
||||
#### How to use
|
||||
|
||||
You can use the model directly after tokenizing the text using the provided tokenizer with the model files.
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
The model is limited to C++ programming language only.
|
||||
|
||||
## Training data
|
||||
|
||||
The model initialized from [`ai-soco-c++-roberta-tiny`](https://github.com/huggingface/transformers/blob/master/model_cards/aliosm/ai-soco-c++-roberta-tiny) model and trained using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset to do text classification.
|
||||
|
||||
## Training procedure
|
||||
|
||||
The model trained on Google Colab platform using V100 GPU for 10 epochs, 32 batch size, 512 max sequence length (sequences larger than 512 were truncated). Each continues 4 spaces were converted to a single tab character (`\t`) before tokenization.
|
||||
|
||||
## Eval results
|
||||
|
||||
The model achieved 87.66%/87.46% accuracy on AI-SOCO task and ranked in the 9th place.
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{ai-soco-2020-fire,
|
||||
title = "Overview of the {PAN@FIRE} 2020 Task on {Authorship Identification of SOurce COde (AI-SOCO)}",
|
||||
author = "Fadel, Ali and Musleh, Husam and Tuffaha, Ibraheem and Al-Ayyoub, Mahmoud and Jararweh, Yaser and Benkhelifa, Elhadj and Rosso, Paolo",
|
||||
booktitle = "Proceedings of The 12th meeting of the Forum for Information Retrieval Evaluation (FIRE 2020)",
|
||||
year = "2020"
|
||||
}
|
||||
```
|
||||
|
||||
<a href="https://huggingface.co/exbert/?model=aliosm/ai-soco-c++-roberta-tiny-clas">
|
||||
<img width="300px" src="https://hf-dinosaur.huggingface.co/exbert/button.png">
|
||||
</a>
|
||||
@@ -0,0 +1,55 @@
|
||||
---
|
||||
language: "c++"
|
||||
tags:
|
||||
- exbert
|
||||
- authorship-identification
|
||||
- fire2020
|
||||
- pan2020
|
||||
- ai-soco
|
||||
license: "mit"
|
||||
datasets:
|
||||
- ai-soco
|
||||
metrics:
|
||||
- perplexity
|
||||
---
|
||||
|
||||
# ai-soco-c++-roberta-tiny
|
||||
|
||||
## Model description
|
||||
|
||||
From scratch pre-trained RoBERTa model with 1 layers and 12 attention heads using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset which consists of C++ codes crawled from CodeForces website.
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
The model can be used to do code classification, authorship identification and other downstream tasks on C++ programming language.
|
||||
|
||||
#### How to use
|
||||
|
||||
You can use the model directly after tokenizing the text using the provided tokenizer with the model files.
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
The model is limited to C++ programming language only.
|
||||
|
||||
## Training data
|
||||
|
||||
The model initialized randomly and trained using [AI-SOCO](https://sites.google.com/view/ai-soco-2020) dataset which contains 100K C++ source codes.
|
||||
|
||||
## Training procedure
|
||||
|
||||
The model trained on Google Colab platform with 8 TPU cores for 200 epochs, 32\*8 batch size, 512 max sequence length and MLM objective. Other parameters were defaulted to the values mentioned in [`run_language_modelling.py`](https://github.com/huggingface/transformers/blob/master/examples/language-modeling/run_language_modeling.py) script. Each continues 4 spaces were converted to a single tab character (`\t`) before tokenization.
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{ai-soco-2020-fire,
|
||||
title = "Overview of the {PAN@FIRE} 2020 Task on {Authorship Identification of SOurce COde (AI-SOCO)}",
|
||||
author = "Fadel, Ali and Musleh, Husam and Tuffaha, Ibraheem and Al-Ayyoub, Mahmoud and Jararweh, Yaser and Benkhelifa, Elhadj and Rosso, Paolo",
|
||||
booktitle = "Proceedings of The 12th meeting of the Forum for Information Retrieval Evaluation (FIRE 2020)",
|
||||
year = "2020"
|
||||
}
|
||||
```
|
||||
|
||||
<a href="https://huggingface.co/exbert/?model=aliosm/ai-soco-c++-roberta-tiny">
|
||||
<img width="300px" src="https://hf-dinosaur.huggingface.co/exbert/button.png">
|
||||
</a>
|
||||
@@ -0,0 +1,52 @@
|
||||
---
|
||||
language:
|
||||
- en
|
||||
tags:
|
||||
- BioNLP
|
||||
- social_media
|
||||
---
|
||||
|
||||
# BioRedditBERT
|
||||
|
||||
## Model description
|
||||
BioRedditBERT is a BERT model initialised from BioBERT (`BioBERT-Base v1.0 + PubMed 200K + PMC 270K`) and further pre-trained on health-related Reddit posts. Please view our paper [COMETA: A Corpus for Medical Entity Linking in the Social Media](https://arxiv.org/pdf/2010.03295.pdf) (EMNLP 2020) for more details.
|
||||
|
||||
|
||||
## Training data
|
||||
|
||||
We crawled all threads from 68 health themed subreddits such as `r/AskDocs`, `r/health` and etc. starting from the beginning of 2015 to the end of 2018, obtaining a collection of more than
|
||||
800K discussions. This collection was then pruned by removing deleted posts, comments from bots or moderators, and so on. In the end, we obtained the training corpus with ca. 300 million tokens and a vocabulary
|
||||
size of ca. 780,000 words.
|
||||
|
||||
## Training procedure
|
||||
We use the same pre-training script in the original [google-research/bert](https://github.com/google-research/bert) repo. The model is initialised with [`BioBERT-Base v1.0 + PubMed 200K + PMC 270K`](https://github.com/dmis-lab/biobert).
|
||||
We train with a batch size of 64, a max sequence length of 64, a learning rate of `2e-5` for 100k steps on two GeForce GTX 1080Ti (11 GB) GPUs. Other hyper-parameters are the same as default.
|
||||
|
||||
|
||||
## Eval results
|
||||
To show the benefit from further pre-training on the social media domain, we demonstrate results on a medical entity linking dataset also in the social media: [AskAPatient](https://zenodo.org/record/55013#.X4ncRmTYpb8) [(Limsopatham and Collier 2016)](https://www.aclweb.org/anthology/P16-1096.pdf).
|
||||
We follow the same 10-fold cross-validation procedure for all models and report the average result without fine-tuning. `[CLS]` is used as representations for entity mentions (we also tried average of all tokens but found `[CLS]` generally performs better).
|
||||
|
||||
Model | Accuracy@1 | Accuracy@5
|
||||
-------|---------|---------
|
||||
[BERT-base-uncased](https://huggingface.co/bert-base-uncased) | 38.2 | 43.3
|
||||
[BioBERT v1.1](https://huggingface.co/dmis-lab/biobert-v1.1) | 41.4 | 51.5
|
||||
[ClinicalBERT](https://huggingface.co/emilyalsentzer/Bio_ClinicalBERT) | 43.9 | 54.3
|
||||
[BlueBERT](https://ftp.ncbi.nlm.nih.gov/pub/lu/Suppl/NCBI-BERT/NCBI_BERT_pubmed_mimic_uncased_L-12_H-768_A-12.zip) | 41.5 | 48.5
|
||||
[SciBERT](https://huggingface.co/allenai/scibert_scivocab_uncased) | 42.3 | 51.9
|
||||
[PubMedBERT](https://huggingface.co/microsoft/BiomedNLP-PubMedBERT-base-uncased-abstract-fulltext) | 42.5 | 49.6
|
||||
BioRedditBERT | **44.3** | **56.2**
|
||||
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{basaldella-2020-cometa,
|
||||
title = "{COMETA}: A Corpus for Medical Entity Linking in the Social Media",
|
||||
author = "Basaldella, Marco and Liu, Fangyu, and Shareghi, Ehsan, and Collier, Nigel",
|
||||
booktitle = "Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing",
|
||||
month = nov,
|
||||
year = "2020",
|
||||
publisher = "Association for Computational Linguistics"
|
||||
}
|
||||
```
|
||||
@@ -1,7 +1,8 @@
|
||||
---
|
||||
language: fr
|
||||
|
||||
license: mit
|
||||
datasets:
|
||||
- oscar
|
||||
---
|
||||
|
||||
# CamemBERT: a Tasty French Language Model
|
||||
|
||||
@@ -2,4 +2,7 @@
|
||||
license: mit
|
||||
thumbnail: https://huggingface.co/front/thumbnails/facebook.png
|
||||
pipeline_tag: zero-shot-classification
|
||||
widget:
|
||||
- text: "Last week I upgraded my iOS version and ever since then my phone has been overheating whenever I use your app."
|
||||
labels: "mobile, website, billing, account access"
|
||||
---
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
---
|
||||
language: ar
|
||||
---
|
||||
# Arabic Named Entity Recognition Model
|
||||
|
||||
Pretrained BERT-based ([arabic-bert-base](https://huggingface.co/asafaya/bert-base-arabic)) Named Entity Recognition model for Arabic.
|
||||
|
||||
The pre-trained model can recognize the following entities:
|
||||
1. **PERSON**
|
||||
|
||||
- و هذا ما نفاه المعاون السياسي للرئيس ***نبيه بري*** ، النائب ***علي حسن خليل***
|
||||
|
||||
- لكن أوساط ***الحريري*** تعتبر أنه ضحى كثيرا في سبيل البلد
|
||||
|
||||
- و ستفقد الملكة ***إليزابيث الثانية*** بذلك سيادتها على واحدة من آخر ممالك الكومنولث
|
||||
|
||||
2. **ORGANIZATION**
|
||||
|
||||
- حسب أرقام ***البنك الدولي***
|
||||
|
||||
- أعلن ***الجيش العراقي***
|
||||
|
||||
- و نقلت وكالة ***رويترز*** عن ثلاثة دبلوماسيين في ***الاتحاد الأوروبي*** ، أن ***بلجيكا*** و ***إيرلندا*** و ***لوكسمبورغ*** تريد أيضاً مناقشة
|
||||
|
||||
- ***الحكومة الاتحادية*** و ***حكومة إقليم كردستان***
|
||||
|
||||
- و هو ما يثير الشكوك حول مشاركة النجم البرتغالي في المباراة المرتقبة أمام ***برشلونة*** الإسباني في
|
||||
|
||||
|
||||
3. ***LOCATION***
|
||||
|
||||
- الجديد هو تمكين اللاجئين من “ مغادرة الجزيرة تدريجياً و بهدوء إلى ***أثينا*** ”
|
||||
|
||||
- ***جزيرة ساكيز*** تبعد 1 كم عن ***إزمير***
|
||||
|
||||
|
||||
4. **DATE**
|
||||
|
||||
- ***غدا الجمعة***
|
||||
|
||||
- ***06 أكتوبر 2020***
|
||||
|
||||
- ***العام السابق***
|
||||
|
||||
|
||||
5. **PRODUCT**
|
||||
|
||||
- عبر حسابه ب ***تطبيق “ إنستغرام ”***
|
||||
|
||||
- الجيل الثاني من ***نظارة الواقع الافتراضي أوكولوس كويست*** تحت اسم " ***أوكولوس كويست 2*** "
|
||||
|
||||
|
||||
6. **COMPETITION**
|
||||
|
||||
- عدم المشاركة في ***بطولة فرنسا المفتوحة للتنس***
|
||||
|
||||
- في مباراة ***كأس السوبر الأوروبي***
|
||||
|
||||
7. **PRIZE**
|
||||
|
||||
- ***جائزة نوبل ل لآداب***
|
||||
|
||||
- الذي فاز ب ***جائزة “ إيمي ” لأفضل دور مساند***
|
||||
|
||||
8. **EVENT**
|
||||
|
||||
- تسجّل أغنية جديدة خاصة ب ***العيد الوطني السعودي***
|
||||
|
||||
- ***مهرجان المرأة يافوية*** في دورته الرابعة
|
||||
|
||||
9. **DISEASE**
|
||||
|
||||
- في مكافحة فيروس ***كورونا*** و عدد من الأمراض
|
||||
|
||||
- الأزمات المشابهة مثل “ ***انفلونزا الطيور*** ” و ” ***انفلونزا الخنازير***
|
||||
|
||||
## Example
|
||||
|
||||
[Find here a complete example to use this model](https://github.com/hatmimoha/arabic-ner)
|
||||
|
||||
Here is the map from index to label:
|
||||
|
||||
```
|
||||
id2label = {
|
||||
"0": "B-PERSON",
|
||||
"1": "I-PERSON",
|
||||
"2": "B-ORGANIZATION",
|
||||
"3": "I-ORGANIZATION",
|
||||
"4": "B-LOCATION",
|
||||
"5": "I-LOCATION",
|
||||
"6": "B-DATE",
|
||||
"7": "I-DATE"",
|
||||
"8": "B-COMPETITION",
|
||||
"9": "I-COMPETITION",
|
||||
"10": "B-PRIZE",
|
||||
"11": "I-PRIZE",
|
||||
"12": "O",
|
||||
"13": "B-PRODUCT",
|
||||
"14": "I-PRODUCT",
|
||||
"15": "B-EVENT",
|
||||
"16": "I-EVENT",
|
||||
"17": "B-DISEASE",
|
||||
"18": "I-DISEASE",
|
||||
}
|
||||
|
||||
```
|
||||
|
||||
## Training Corpus
|
||||
|
||||
The training corpus is made of 378.000 tokens (14.000 sentences) collected from the Web and annotated manually.
|
||||
|
||||
## Results
|
||||
|
||||
The results on a valid corpus made of 30.000 tokens shows an F-measure of ~87%.
|
||||
@@ -0,0 +1,9 @@
|
||||
# DynaBERT: Dynamic BERT with Adaptive Width and Depth
|
||||
|
||||
* DynaBERT can flexibly adjust the size and latency by selecting adaptive width and depth, and
|
||||
the subnetworks of it have competitive performances as other similar-sized compressed models.
|
||||
The training process of DynaBERT includes first training a width-adaptive BERT and then
|
||||
allowing both adaptive width and depth using knowledge distillation.
|
||||
|
||||
* This code is modified based on the repository developed by Hugging Face: [Transformers v2.1.1](https://github.com/huggingface/transformers/tree/v2.1.1)
|
||||
* The results in the paper are produced by using single V100 GPU.
|
||||
@@ -1,5 +1,11 @@
|
||||
---
|
||||
language: fr
|
||||
tags:
|
||||
- question-answering
|
||||
- camembert
|
||||
license: gpl-3.0
|
||||
datasets:
|
||||
- fquad
|
||||
---
|
||||
|
||||
# camembert-base-fquad
|
||||
|
||||
@@ -1,5 +1,11 @@
|
||||
---
|
||||
language: fr
|
||||
tags:
|
||||
- question-answering
|
||||
- camembert
|
||||
license: gpl-3.0
|
||||
datasets:
|
||||
- fquad
|
||||
---
|
||||
|
||||
# camembert-large-fquad
|
||||
|
||||
@@ -4,6 +4,9 @@ thumbnail: https://miro.medium.com/max/700/1*MoPnD6vA9wTHjdLfW7POyw.png
|
||||
widget:
|
||||
- text: "Le camembert LePetit c'est le <mask>."
|
||||
- text: "Salut les <mask> ça va ?"
|
||||
license: gpl-3.0
|
||||
tags:
|
||||
- masked-lm
|
||||
---
|
||||
|
||||
# LePetit: A pre-training efficient and lightning fast French Language Model
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Cased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Cased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Uncased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Uncased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Cased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Cased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Uncased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Uncased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -5,8 +5,7 @@ tags:
|
||||
- pytorch
|
||||
datasets:
|
||||
- yahoo-answers
|
||||
widget:
|
||||
- text: "Who are you voting for in 2020? <sep> This text is about politics."
|
||||
pipeline_tag: zero-shot-classification
|
||||
---
|
||||
|
||||
# bart-lage-mnli-yahoo-answers
|
||||
|
||||
@@ -5,11 +5,17 @@ tags:
|
||||
- pytorch
|
||||
- tensorflow
|
||||
datasets:
|
||||
- mnli
|
||||
- multi_nli
|
||||
- xnli
|
||||
widget:
|
||||
- text: "За кого вы голосуете в 2020 году? <sep> This text is about politique."
|
||||
license: mit
|
||||
pipeline_tag: zero-shot-classification
|
||||
widget:
|
||||
- text: "За кого вы голосуете в 2020 году?"
|
||||
labels: "politique étrangère, Europe, élections, affaires, politique"
|
||||
- text: "لمن تصوت في 2020؟"
|
||||
labels: "السياسة الخارجية, أوروبا, الانتخابات, الأعمال, السياسة"
|
||||
- text: "2020'de kime oy vereceksiniz?"
|
||||
labels: "dış politika, Avrupa, seçimler, ticaret, siyaset"
|
||||
---
|
||||
|
||||
# xlm-roberta-large-xnli
|
||||
@@ -115,4 +121,3 @@ This model was pre-trained on set of 100 languages, as described in
|
||||
MNLI train set and the XNLI validation and test sets. Finally, it was trained for one additional epoch on only XNLI
|
||||
data where the translations for the premise and hypothesis are shuffled such that the premise and hypothesis for
|
||||
each example come from the same original English example but the premise and hypothesis are of different languages.
|
||||
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
---
|
||||
language:
|
||||
- en
|
||||
- ar
|
||||
datasets:
|
||||
- gigaword
|
||||
- oscar
|
||||
- wikipedia
|
||||
---
|
||||
|
||||
## GigaBERT-v3
|
||||
GigaBERT-v3 is a customized bilingual BERT for English and Arabic. It was pre-trained in a large-scale corpus (Gigaword+Oscar+Wikipedia) with ~10B tokens, showing state-of-the-art zero-shot transfer performance from English to Arabic on information extraction (IE) tasks. More details can be found in the following paper:
|
||||
|
||||
@inproceedings{lan2020gigabert,
|
||||
author = {Lan, Wuwei and Chen, Yang and Xu, Wei and Ritter, Alan},
|
||||
title = {GigaBERT: Zero-shot Transfer Learning from English to Arabic},
|
||||
booktitle = {Proceedings of The 2020 Conference on Empirical Methods on Natural Language Processing (EMNLP)},
|
||||
year = {2020}
|
||||
}
|
||||
|
||||
## Usage
|
||||
```
|
||||
from transformers import *
|
||||
tokenizer = BertTokenizer.from_pretrained("lanwuwei/GigaBERT-v3-Arabic-and-English", do_lower_case=True)
|
||||
model = BertForTokenClassification.from_pretrained("lanwuwei/GigaBERT-v3-Arabic-and-English")
|
||||
```
|
||||
More code examples can be found [here](https://github.com/lanwuwei/GigaBERT).
|
||||
@@ -1,3 +1,9 @@
|
||||
---
|
||||
language: en
|
||||
datasets:
|
||||
- cnn_dailymail
|
||||
---
|
||||
|
||||
## prophetnet-large-uncased-cnndm
|
||||
Fine-tuned weights(converted from [original fairseq version repo](https://github.com/microsoft/ProphetNet)) for [ProphetNet](https://arxiv.org/abs/2001.04063) on summarization task CNN/DailyMail.
|
||||
ProphetNet is a new pre-trained language model for sequence-to-sequence learning with a novel self-supervised objective called future n-gram prediction.
|
||||
@@ -15,8 +21,11 @@ inputs = tokenizer([ARTICLE_TO_SUMMARIZE], max_length=100, return_tensors='pt')
|
||||
|
||||
# Generate Summary
|
||||
summary_ids = model.generate(inputs['input_ids'], num_beams=4, max_length=512, early_stopping=True)
|
||||
tokenizer.batch_decode(summary_ids.tolist())
|
||||
tokenizer.batch_decode(summary_ids, skip_special_tokens=True)
|
||||
|
||||
# should give: 'ustc was founded in beijing by the chinese academy of sciences in 1958. [X_SEP] ustc\'s mission was to develop a high - level science and technology workforce. [X_SEP] the establishment was hailed as " a major event in the history of chinese education and science "'
|
||||
```
|
||||
|
||||
Here, [X_SEP] is used as a special token to seperate sentences.
|
||||
### Citation
|
||||
```bibtex
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
language: en
|
||||
datasets:
|
||||
- squad
|
||||
---
|
||||
|
||||
##
|
||||
prophetnet-large-uncased-squad-qg
|
||||
Fine-tuned weights(converted from [original fairseq version repo](https://github.com/microsoft/ProphetNet)) for [ProphetNet](https://arxiv.org/abs/2001.04063) on question generation
|
||||
SQuAD 1.1.
|
||||
ProphetNet is a new pre-trained language model for sequence-to-sequence learning with a novel self-supervised objective called future n-gram prediction.
|
||||
ProphetNet is able to predict more future tokens with a n-stream decoder. The original implementation is Fairseq version at [github repo](https://github.com/microsoft/ProphetNet).
|
||||
|
||||
### Usage
|
||||
```
|
||||
from transformers import ProphetNetTokenizer, ProphetNetForConditionalGeneration, ProphetNetConfig
|
||||
|
||||
model = ProphetNetForConditionalGeneration.from_pretrained('microsoft/prophetnet-large-uncased-squad-qg')
|
||||
tokenizer = ProphetNetTokenizer.from_pretrained('microsoft/prophetnet-large-uncased-squad-qg')
|
||||
|
||||
FACT_TO_GENERATE_QUESTION_FROM = ""Bill Gates [SEP] Microsoft was founded by Bill Gates and Paul Allen on April 4, 1975."
|
||||
|
||||
inputs = tokenizer([FACT_TO_GENERATE_QUESTION_FROM], return_tensors='pt')
|
||||
|
||||
# Generate Summary
|
||||
question_ids = model.generate(inputs['input_ids'], num_beams=5, early_stopping=True)
|
||||
tokenizer.batch_decode(question_ids, skip_special_tokens=True)
|
||||
|
||||
# should give: 'along with paul allen, who founded microsoft?'
|
||||
```
|
||||
### Citation
|
||||
```bibtex
|
||||
@article{yan2020prophetnet,
|
||||
title={Prophetnet: Predicting future n-gram for sequence-to-sequence pre-training},
|
||||
author={Yan, Yu and Qi, Weizhen and Gong, Yeyun and Liu, Dayiheng and Duan, Nan and Chen, Jiusheng and Zhang, Ruofei and Zhou, Ming},
|
||||
journal={arXiv preprint arXiv:2001.04063},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
@@ -1,10 +1,30 @@
|
||||
---
|
||||
language: en
|
||||
---
|
||||
|
||||
## prophetnet-large-uncased
|
||||
Pretrained weights for [ProphetNet](https://arxiv.org/abs/2001.04063).
|
||||
ProphetNet is a new pre-trained language model for sequence-to-sequence learning with a novel self-supervised objective called future n-gram prediction.
|
||||
ProphetNet is able to predict more future tokens with a n-stream decoder. The original implementation is Fairseq version at [github repo](https://github.com/microsoft/ProphetNet).
|
||||
|
||||
### Usage
|
||||
Please see [the official repository](https://github.com/microsoft/ProphetNet) for details.
|
||||
|
||||
This pre-trained model can be fine-tuned on *sequence-to-sequence* tasks. The model could *e.g.* be trained on headline generation as follows:
|
||||
|
||||
```python
|
||||
from transformers import ProphetNetForConditionalGeneration, ProphetNetTokenizer
|
||||
|
||||
model = ProphetNetForConditionalGeneration.from_pretrained("microsoft/prophetnet-large-uncased")
|
||||
tokenizer = ProphetNetTokenizer.from_pretrained("microsoft/prophetnet-large-uncased")
|
||||
|
||||
input_str = "the us state department said wednesday it had received no formal word from bolivia that it was expelling the us ambassador there but said the charges made against him are `` baseless ."
|
||||
target_str = "us rejects charges against its ambassador in bolivia"
|
||||
|
||||
input_ids = tokenizer(input_str, return_tensors="pt").input_ids
|
||||
labels = tokenizer(target_str, return_tensors="pt").input_ids
|
||||
|
||||
loss = model(input_ids, labels=labels, return_dict=True).loss
|
||||
```
|
||||
|
||||
### Citation
|
||||
```bibtex
|
||||
|
||||
@@ -8,7 +8,7 @@ For xGLUE corss-lingual NLG tasks, xProphetNet is finetuned with English data, b
|
||||
### Usage
|
||||
A quick usage is like:
|
||||
```
|
||||
from transformers import ProphetNetTokenizer, ProphetNetForConditionalGeneration, ProphetNetConfig
|
||||
from transformers import XLMProphetNetTokenizer, XLMProphetNetForConditionalGeneration, ProphetNetConfig
|
||||
|
||||
model = ProphetNetForConditionalGeneration.from_pretrained('microsoft/xprophetnet-large-wiki100-cased-xglue-ntg')
|
||||
tokenizer = ProphetNetTokenizer.from_pretrained('microsoft/xprophetnet-large-wiki100-cased-xglue-ntg')
|
||||
@@ -19,7 +19,12 @@ ZH_SENTENCE = "根据该组织的官方门户网站,微软公司打算在2020
|
||||
inputs = tokenizer([EN_SENTENCE, RU_SENTENCE, ZH_SENTENCE], padding=True, max_length=256, return_tensors='pt')
|
||||
|
||||
summary_ids = model.generate(inputs['input_ids'], num_beams=4, max_length=100, early_stopping=True)
|
||||
print([tokenizer.decode(g) for g in summary_ids])
|
||||
tokenizer.batch_decode(summary_ids, skip_special_tokens=True)
|
||||
|
||||
# should give:
|
||||
# 'Microsoft to end Windows 7 free support after January 14, 2020'
|
||||
# 'Microsoft намерена прекратить бесплатную поддержку Windows 7 после 14 января 2020 года'
|
||||
# '微软终止对Windows 7操作系统的免费支持'
|
||||
```
|
||||
### Citation
|
||||
```bibtex
|
||||
|
||||
@@ -1,12 +1,35 @@
|
||||
---
|
||||
language: multilingual
|
||||
---
|
||||
|
||||
## xprophetnet-large-wiki100-cased
|
||||
Cross-lingual version [ProphetNet](https://arxiv.org/abs/2001.04063), pretrained on [wiki100 xGLUE dataset](https://arxiv.org/abs/2004.01401).
|
||||
ProphetNet is a new pre-trained language model for sequence-to-sequence learning with a novel self-supervised objective called future n-gram prediction.
|
||||
ProphetNet is able to predict more future tokens with a n-stream decoder. The original implementation is Fairseq version at [github repo](https://github.com/microsoft/ProphetNet).
|
||||
|
||||
xProphetNet is also served as the baseline model for xGLUE cross-lingual natural language generation tasks.
|
||||
For xGLUE corss-lingual NLG tasks, xProphetNet is finetuned with English data, but inference with both English and other zero-shot language data.
|
||||
For xGLUE corss-lingual NLG tasks, xProphetNet is finetuned with English data, but inference with both English and other zero-shot language data.
|
||||
|
||||
### Usage
|
||||
Please see [the official repository](https://github.com/microsoft/ProphetNet/tree/master/xProphetNet) for details.
|
||||
|
||||
This pre-trained model can be fine-tuned on *sequence-to-sequence* tasks. The model could *e.g.* be trained on English headline generation as follows:
|
||||
|
||||
```python
|
||||
from transformers import XLMProphetNetForConditionalGeneration, XLMProphetNetTokenizer
|
||||
|
||||
model = XLMProphetNetForConditionalGeneration.from_pretrained("microsoft/xprophetnet-large-wiki100-cased")
|
||||
tokenizer = XLMProphetNetTokenizer.from_pretrained("microsoft/xprophetnet-large-wiki100-cased")
|
||||
|
||||
input_str = "the us state department said wednesday it had received no formal word from bolivia that it was expelling the us ambassador there but said the charges made against him are `` baseless ."
|
||||
target_str = "us rejects charges against its ambassador in bolivia"
|
||||
|
||||
input_ids = tokenizer(input_str, return_tensors="pt").input_ids
|
||||
labels = tokenizer(target_str, return_tensors="pt").input_ids
|
||||
|
||||
loss = model(input_ids, labels=labels, return_dict=True).loss
|
||||
```
|
||||
|
||||
Note that since this model is a multi-lingual model it can be fine-tuned on all kinds of other languages.
|
||||
|
||||
### Citation
|
||||
```bibtex
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
---
|
||||
language: de
|
||||
---
|
||||
|
||||
# German BERT + LER (Legal Entity Recognition) ⚖️
|
||||
|
||||
German BERT ([BERT-base-german-cased](https://huggingface.co/bert-base-german-cased)) fine-tuned on [Legal-Entity-Recognition](https://github.com/elenanereiss/Legal-Entity-Recognition) dataset for **LER** (NER) downstream task.
|
||||
|
||||
## Details of the downstream task (NER) - Dataset
|
||||
|
||||
[Legal-Entity-Recognition](https://github.com/elenanereiss/Legal-Entity-Recognition): Fine-grained Named Entity Recognition in Legal Documents.
|
||||
|
||||
Court decisions from 2017 and 2018 were selected for the dataset, published online by the [Federal Ministry of Justice and Consumer Protection](http://www.rechtsprechung-im-internet.de). The documents originate from seven federal courts: Federal Labour Court (BAG), Federal Fiscal Court (BFH), Federal Court of Justice (BGH), Federal Patent Court (BPatG), Federal Social Court (BSG), Federal Constitutional Court (BVerfG) and Federal Administrative Court (BVerwG).
|
||||
|
||||
|
||||
| Split | # Samples |
|
||||
| ---------------------- | ----- |
|
||||
| Train | 1657048 |
|
||||
| Eval | 500000 |
|
||||
|
||||
- Training script: [Fine-tuning script for NER provided by Huggingface](https://github.com/huggingface/transformers/blob/master/examples/token-classification/run_ner.py)
|
||||
Colab: [How to fine-tune a model for NER using HF scripts](https://colab.research.google.com/drive/156Qrd7NsUHwA3nmQ6gXdZY0NzOvqk9AT?usp=sharing)
|
||||
|
||||
- Labels covered (and its distribution):
|
||||
|
||||
```
|
||||
107 B-AN
|
||||
918 B-EUN
|
||||
2238 B-GRT
|
||||
13282 B-GS
|
||||
1113 B-INN
|
||||
704 B-LD
|
||||
151 B-LDS
|
||||
2490 B-LIT
|
||||
282 B-MRK
|
||||
890 B-ORG
|
||||
1374 B-PER
|
||||
1480 B-RR
|
||||
10046 B-RS
|
||||
401 B-ST
|
||||
68 B-STR
|
||||
1011 B-UN
|
||||
282 B-VO
|
||||
391 B-VS
|
||||
2648 B-VT
|
||||
46 I-AN
|
||||
6925 I-EUN
|
||||
1957 I-GRT
|
||||
70257 I-GS
|
||||
2931 I-INN
|
||||
153 I-LD
|
||||
26 I-LDS
|
||||
28881 I-LIT
|
||||
383 I-MRK
|
||||
1185 I-ORG
|
||||
330 I-PER
|
||||
106 I-RR
|
||||
138938 I-RS
|
||||
34 I-ST
|
||||
55 I-STR
|
||||
1259 I-UN
|
||||
1572 I-VO
|
||||
2488 I-VS
|
||||
11121 I-VT
|
||||
1348525 O
|
||||
```
|
||||
- [Annotation Guidelines (German)](https://github.com/elenanereiss/Legal-Entity-Recognition/blob/master/docs/Annotationsrichtlinien.pdf)
|
||||
|
||||
|
||||
## Metrics on evaluation set
|
||||
|
||||
| Metric | # score |
|
||||
| :------------------------------------------------------------------------------------: | :-------: |
|
||||
| F1 | **85.67**
|
||||
| Precision | **84.35** |
|
||||
| Recall | **87.04** |
|
||||
| Accuracy | **98.46** |
|
||||
|
||||
## Model in action
|
||||
|
||||
Fast usage with **pipelines**:
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
nlp_ler = pipeline(
|
||||
"ner",
|
||||
model="mrm8488/bert-base-german-finetuned-ler",
|
||||
tokenizer="mrm8488/bert-base-german-finetuned-ler"
|
||||
)
|
||||
|
||||
text = "Your German legal text here"
|
||||
|
||||
nlp_ler(text)
|
||||
|
||||
```
|
||||
|
||||
> Created by [Manuel Romero/@mrm8488](https://twitter.com/mrm8488)
|
||||
|
||||
> Made with <span style="color: #e25555;">♥</span> in Spain
|
||||
@@ -0,0 +1,79 @@
|
||||
---
|
||||
language: en
|
||||
datasets:
|
||||
- common_gen
|
||||
---
|
||||
|
||||
# T5-base fine-tuned on CommonGen
|
||||
|
||||
[Google's T5](https://ai.googleblog.com/2020/02/exploring-transfer-learning-with-t5.html) fine-tuned on [CommonGen](https://inklab.usc.edu/CommonGen/index.html) for *Generative Commonsense Reasoning*.
|
||||
|
||||
## Details of T5
|
||||
|
||||
The **T5** model was presented in [Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer](https://arxiv.org/pdf/1910.10683.pdf) by *Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, Peter J. Liu* in Here the abstract:
|
||||
|
||||
Transfer learning, where a model is first pre-trained on a data-rich task before being fine-tuned on a downstream task, has emerged as a powerful technique in natural language processing (NLP). The effectiveness of transfer learning has given rise to a diversity of approaches, methodology, and practice. In this paper, we explore the landscape of transfer learning techniques for NLP by introducing a unified framework that converts every language problem into a text-to-text format. Our systematic study compares pre-training objectives, architectures, unlabeled datasets, transfer approaches, and other factors on dozens of language understanding tasks. By combining the insights from our exploration with scale and our new “Colossal Clean Crawled Corpus”, we achieve state-of-the-art results on many benchmarks covering summarization, question answering, text classification, and more. To facilitate future work on transfer learning for NLP, we release our dataset, pre-trained models, and code.
|
||||
|
||||

|
||||
|
||||
|
||||
## Details of the dataset 📚
|
||||
|
||||
CommonGen is a constrained text generation task, associated with a benchmark dataset, to explicitly test machines for the ability of generative commonsense reasoning. Given a set of common concepts; the task is to generate a coherent sentence describing an everyday scenario using these concepts.
|
||||
|
||||
CommonGen is challenging because it inherently requires 1) relational reasoning using background commonsense knowledge, and 2) compositional generalization ability to work on unseen concept combinations. Our dataset, constructed through a combination of crowd-sourcing from AMT and existing caption corpora, consists of 30k concept-sets and 50k sentences in total.
|
||||
|
||||
|
||||
| Dataset | Split | # samples |
|
||||
| -------- | ----- | --------- |
|
||||
| common_gen | train | 67389 |
|
||||
| common_gen | valid | 4018 |
|
||||
| common_gen | test | 1497 |
|
||||
|
||||
|
||||
|
||||
## Model fine-tuning 🏋️
|
||||
|
||||
The training script is a slightly modified version of [this awesome one](https://colab.research.google.com/github/patil-suraj/exploring-T5/blob/master/T5_on_TPU.ipynb) by [Suraj Patil](https://twitter.com/psuraj28)
|
||||
|
||||
## Metrics 📋
|
||||
|
||||
| Metric | Score |
|
||||
|--------|-------|
|
||||
|ROUGE-2 | 17.10 |
|
||||
|ROUGE-L | 39.47 |
|
||||
|BLEU | WIP |
|
||||
|
||||
The metrics above slightly improves results shown in the [paper](https://arxiv.org/abs/1911.03705) for the same model and metrics.
|
||||
|
||||
|
||||
## Model in Action 🚀
|
||||
|
||||
```python
|
||||
from transformers import AutoModelWithLMHead, AutoTokenizer
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("mrm8488/t5-base-finetuned-common_gen")
|
||||
model = AutoModelWithLMHead.from_pretrained("mrm8488/t5-base-finetuned-common_gen")
|
||||
|
||||
def gen_sentence(words, max_length=32):
|
||||
input_text = words
|
||||
features = tokenizer([input_text], return_tensors='pt')
|
||||
|
||||
output = model.generate(input_ids=features['input_ids'],
|
||||
attention_mask=features['attention_mask'],
|
||||
max_length=max_length)
|
||||
|
||||
return tokenizer.decode(output[0])
|
||||
|
||||
words = "tree plant ground hole dig"
|
||||
|
||||
gen_sentence(words)
|
||||
|
||||
# output: digging a hole in the ground to plant trees
|
||||
```
|
||||
[](https://colab.research.google.com/github/mrm8488/shared_colab_notebooks/blob/master/T5_base_finetuned_common_gen.ipynb)
|
||||
|
||||
|
||||
> Created by [Manuel Romero/@mrm8488](https://twitter.com/mrm8488) | [LinkedIn](https://www.linkedin.com/in/manuel-romero-cs/)
|
||||
|
||||
> Made with <span style="color: #e25555;">♥</span> in Spain
|
||||
@@ -29,7 +29,7 @@ Dataset ID: ```squad``` from [HugginFace/NLP](https://github.com/huggingface/nl
|
||||
How to load it from [nlp](https://github.com/huggingface/nlp)
|
||||
|
||||
```python
|
||||
train_dataset = nlp.load_dataset('squad, split=nlp.Split.TRAIN)
|
||||
train_dataset = nlp.load_dataset('squad', split=nlp.Split.TRAIN)
|
||||
valid_dataset = nlp.load_dataset('squad', split=nlp.Split.VALIDATION)
|
||||
```
|
||||
Check out more about this dataset and others in [NLP Viewer](https://huggingface.co/nlp/viewer/)
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
---
|
||||
language: it
|
||||
datasets:
|
||||
- xtreme
|
||||
---
|
||||
|
||||
# Italian-Bert (Italian Bert) + POS 🎃🏷
|
||||
|
||||
This model is a fine-tuned on [xtreme udpos Italian](https://huggingface.co/nlp/viewer/?dataset=xtreme&config=udpos.Italian) version of [Bert Base Italian](https://huggingface.co/dbmdz/bert-base-italian-cased) for **POS** downstream task.
|
||||
|
||||
## Details of the downstream task (POS) - Dataset
|
||||
|
||||
- [Dataset: xtreme udpos Italian](https://huggingface.co/nlp/viewer/?dataset=xtreme&config=udpos.Italian) 📚
|
||||
|
||||
| Dataset | # Examples |
|
||||
| ---------------------- | ----- |
|
||||
| Train | 716 K |
|
||||
| Dev | 85 K |
|
||||
|
||||
- [Fine-tune on NER script provided by @stefan-it](https://raw.githubusercontent.com/stefan-it/fine-tuned-berts-seq/master/scripts/preprocess.py)
|
||||
|
||||
- Labels covered:
|
||||
|
||||
```
|
||||
ADJ
|
||||
ADP
|
||||
ADV
|
||||
AUX
|
||||
CCONJ
|
||||
DET
|
||||
INTJ
|
||||
NOUN
|
||||
NUM
|
||||
PART
|
||||
PRON
|
||||
PROPN
|
||||
PUNCT
|
||||
SCONJ
|
||||
SYM
|
||||
VERB
|
||||
X
|
||||
```
|
||||
|
||||
## Metrics on evaluation set 🧾
|
||||
|
||||
| Metric | # score |
|
||||
| :------------------------------------------------------------------------------------: | :-------: |
|
||||
| F1 | **97.25**
|
||||
| Precision | **97.15** |
|
||||
| Recall | **97.36** |
|
||||
|
||||
## Model in action 🔨
|
||||
|
||||
|
||||
Example of usage
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
nlp_pos = pipeline(
|
||||
"ner",
|
||||
model="sachaarbonel/bert-italian-cased-finetuned-pos",
|
||||
tokenizer=(
|
||||
'sachaarbonel/bert-spanish-cased-finetuned-pos',
|
||||
{"use_fast": False}
|
||||
))
|
||||
|
||||
|
||||
text = 'Roma è la Capitale d'Italia.'
|
||||
|
||||
nlp_pos(text)
|
||||
|
||||
'''
|
||||
Output:
|
||||
--------
|
||||
[{'entity': 'PROPN', 'index': 1, 'score': 0.9995346665382385, 'word': 'roma'},
|
||||
{'entity': 'AUX', 'index': 2, 'score': 0.9966597557067871, 'word': 'e'},
|
||||
{'entity': 'DET', 'index': 3, 'score': 0.9994786977767944, 'word': 'la'},
|
||||
{'entity': 'NOUN',
|
||||
'index': 4,
|
||||
'score': 0.9995198249816895,
|
||||
'word': 'capitale'},
|
||||
{'entity': 'ADP', 'index': 5, 'score': 0.9990894198417664, 'word': 'd'},
|
||||
{'entity': 'PART', 'index': 6, 'score': 0.57159024477005, 'word': "'"},
|
||||
{'entity': 'PROPN',
|
||||
'index': 7,
|
||||
'score': 0.9994804263114929,
|
||||
'word': 'italia'},
|
||||
{'entity': 'PUNCT', 'index': 8, 'score': 0.9772886633872986, 'word': '.'}]
|
||||
'''
|
||||
```
|
||||
Yeah! Not too bad 🎉
|
||||
|
||||
> Created by [Sacha Arbonel/@sachaarbonel](https://twitter.com/sachaarbonel) | [LinkedIn](https://www.linkedin.com/in/sacha-arbonel)
|
||||
|
||||
> Made with <span style="color: #e25555;">♥</span> in Paris
|
||||
@@ -7,9 +7,22 @@ tags:
|
||||
- translation
|
||||
|
||||
license: apache-2.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
[Google's T5](https://ai.googleblog.com/2020/02/exploring-transfer-learning-with-t5.html)
|
||||
## Disclaimer
|
||||
|
||||
Due do it's immense size, `t5-11b` requires some special treatment.
|
||||
First, `t5-11b` should be loaded with flag `use_cdn` set to `False` as follows:
|
||||
|
||||
```python
|
||||
t5 = transformers.T5ForConditionalGeneration.from_pretrained('t5-11b', use_cdn = False)
|
||||
```
|
||||
|
||||
Secondly, a single GPU will most likely not have enough memory to even load the model into memory as the weights alone amount to over 40 GB.
|
||||
Model parallelism has to be used here to overcome this problem as is explained in this [PR](https://github.com/huggingface/transformers/pull/3578).
|
||||
|
||||
## [Google's T5](https://ai.googleblog.com/2020/02/exploring-transfer-learning-with-t5.html)
|
||||
|
||||
Pretraining Dataset: [C4](https://huggingface.co/datasets/c4)
|
||||
|
||||
@@ -25,14 +38,3 @@ Transfer learning, where a model is first pre-trained on a data-rich task before
|
||||
|
||||

|
||||
|
||||
## Disclaimer
|
||||
|
||||
Due do it's immense size, `t5-11b` requires some special treatment.
|
||||
First, `t5-11b` should be loaded with flag `use_cdn` set to `False` as follows:
|
||||
|
||||
```python
|
||||
t5 = transformers.T5ForConditionalGeneration.from_pretrained('t5-11b', use_cdn = False)
|
||||
```
|
||||
|
||||
Secondly, a single GPU will most likely not have enough memory to even load the model into memory as the weights alone amount to over 40 GB.
|
||||
Model parallelism has to be used here to overcome this problem as is explained in this [PR](https://github.com/huggingface/transformers/pull/3578).
|
||||
|
||||
@@ -4,6 +4,7 @@ datasets:
|
||||
tags:
|
||||
- distilbart
|
||||
- distilbart-mnli
|
||||
pipeline_tag: zero-shot-classification
|
||||
---
|
||||
|
||||
# DistilBart-MNLI
|
||||
|
||||
@@ -4,6 +4,7 @@ datasets:
|
||||
tags:
|
||||
- distilbart
|
||||
- distilbart-mnli
|
||||
pipeline_tag: zero-shot-classification
|
||||
---
|
||||
|
||||
# DistilBart-MNLI
|
||||
|
||||
@@ -4,6 +4,7 @@ datasets:
|
||||
tags:
|
||||
- distilbart
|
||||
- distilbart-mnli
|
||||
pipeline_tag: zero-shot-classification
|
||||
---
|
||||
|
||||
# DistilBart-MNLI
|
||||
|
||||
@@ -4,6 +4,7 @@ datasets:
|
||||
tags:
|
||||
- distilbart
|
||||
- distilbart-mnli
|
||||
pipeline_tag: zero-shot-classification
|
||||
---
|
||||
|
||||
# DistilBart-MNLI
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
---
|
||||
datasets:
|
||||
- snli
|
||||
- anli
|
||||
- multi_nli
|
||||
- multi_nli_mismatch
|
||||
- fever
|
||||
license: mit
|
||||
---
|
||||
This is a strong pre-trained RoBERTa-Large NLI model.
|
||||
|
||||
The training data is a combination of well-known NLI datasets: [`SNLI`](https://nlp.stanford.edu/projects/snli/), [`MNLI`](https://cims.nyu.edu/~sbowman/multinli/), [`FEVER-NLI`](https://github.com/easonnie/combine-FEVER-NSMN/blob/master/other_resources/nli_fever.md), [`ANLI (R1, R2, R3)`](https://github.com/facebookresearch/anli).
|
||||
Other pre-trained NLI models including `RoBERTa`, `ALBert`, `BART`, `ELECTRA`, `XLNet` are also available.
|
||||
|
||||
Trained by [Yixin Nie](https://easonnie.github.io), [original source](https://github.com/facebookresearch/anli).
|
||||
|
||||
Try the code snippet below.
|
||||
```
|
||||
from transformers import AutoTokenizer, AutoModelForSequenceClassification
|
||||
import torch
|
||||
|
||||
if __name__ == '__main__':
|
||||
max_length = 256
|
||||
|
||||
premise = "Two women are embracing while holding to go packages."
|
||||
hypothesis = "The men are fighting outside a deli."
|
||||
|
||||
hg_model_hub_name = "ynie/roberta-large-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/albert-xxlarge-v2-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/bart-large-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/electra-large-discriminator-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/xlnet-large-cased-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(hg_model_hub_name)
|
||||
model = AutoModelForSequenceClassification.from_pretrained(hg_model_hub_name)
|
||||
|
||||
tokenized_input_seq_pair = tokenizer.encode_plus(premise, hypothesis,
|
||||
max_length=max_length,
|
||||
return_token_type_ids=True, truncation=True)
|
||||
|
||||
input_ids = torch.Tensor(tokenized_input_seq_pair['input_ids']).long().unsqueeze(0)
|
||||
# remember bart doesn't have 'token_type_ids', remove the line below if you are using bart.
|
||||
token_type_ids = torch.Tensor(tokenized_input_seq_pair['token_type_ids']).long().unsqueeze(0)
|
||||
attention_mask = torch.Tensor(tokenized_input_seq_pair['attention_mask']).long().unsqueeze(0)
|
||||
|
||||
outputs = model(input_ids,
|
||||
attention_mask=attention_mask,
|
||||
token_type_ids=token_type_ids,
|
||||
labels=None)
|
||||
# Note:
|
||||
# "id2label": {
|
||||
# "0": "entailment",
|
||||
# "1": "neutral",
|
||||
# "2": "contradiction"
|
||||
# },
|
||||
|
||||
predicted_probability = torch.softmax(outputs[0], dim=1)[0].tolist() # batch_size only one
|
||||
|
||||
print("Premise:", premise)
|
||||
print("Hypothesis:", hypothesis)
|
||||
print("Entailment:", predicted_probability[0])
|
||||
print("Neutral:", predicted_probability[1])
|
||||
print("Contradiction:", predicted_probability[2])
|
||||
```
|
||||
|
||||
More in [here](https://github.com/facebookresearch/anli/blob/master/src/hg_api/interactive_eval.py).
|
||||
|
||||
Citation:
|
||||
```
|
||||
@inproceedings{nie-etal-2020-adversarial,
|
||||
title = "Adversarial {NLI}: A New Benchmark for Natural Language Understanding",
|
||||
author = "Nie, Yixin and
|
||||
Williams, Adina and
|
||||
Dinan, Emily and
|
||||
Bansal, Mohit and
|
||||
Weston, Jason and
|
||||
Kiela, Douwe",
|
||||
booktitle = "Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics",
|
||||
year = "2020",
|
||||
publisher = "Association for Computational Linguistics",
|
||||
}
|
||||
```
|
||||
+3
-1
@@ -48,4 +48,6 @@ Pull Request so it can be included under the Community notebooks.
|
||||
|[fine-tune a non-English GPT-2 Model with Trainer class](https://github.com/philschmid/fine-tune-GPT-2/blob/master/Fine_tune_a_non_English_GPT_2_Model_with_Huggingface.ipynb) | How to fine-tune a non-English GPT-2 Model with Trainer class | [Philipp Schmid](https://www.philschmid.de) | [](https://colab.research.google.com/github/philschmid/fine-tune-GPT-2/blob/master/Fine_tune_a_non_English_GPT_2_Model_with_Huggingface.ipynb)|
|
||||
|[Fine-tune a DistilBERT Model for Multi Label Classification task](https://github.com/DhavalTaunk08/Transformers_scripts/blob/master/Transformers_multilabel_distilbert.ipynb) | How to fine-tune a DistilBERT Model for Multi Label Classification task | [Dhaval Taunk](https://github.com/DhavalTaunk08) | [](https://colab.research.google.com/github/DhavalTaunk08/Transformers_scripts/blob/master/Transformers_multilabel_distilbert.ipynb)|
|
||||
|[Fine-tune ALBERT for sentence-pair classification](https://github.com/NadirEM/nlp-notebooks/blob/master/Fine_tune_ALBERT_sentence_pair_classification.ipynb) | How to fine-tune an ALBERT model or another BERT-based model for the sentence-pair classification task | [Nadir El Manouzi](https://github.com/NadirEM) | [](https://colab.research.google.com/github/NadirEM/nlp-notebooks/blob/master/Fine_tune_ALBERT_sentence_pair_classification.ipynb)|
|
||||
|[Fine-tune Roberta for sentiment analysis](https://github.com/DhavalTaunk08/NLP_scripts/blob/master/sentiment_analysis_using_roberta.ipynb) | How to fine-tune an Roberta model for sentiment analysis | [Dhaval Taunk](https://github.com/DhavalTaunk08) | [](https://colab.research.google.com/github/DhavalTaunk08/NLP_scripts/blob/master/sentiment_analysis_using_roberta.ipynb)|
|
||||
|[Fine-tune Roberta for sentiment analysis](https://github.com/DhavalTaunk08/NLP_scripts/blob/master/sentiment_analysis_using_roberta.ipynb) | How to fine-tune an Roberta model for sentiment analysis | [Dhaval Taunk](https://github.com/DhavalTaunk08) | [](https://colab.research.google.com/github/DhavalTaunk08/NLP_scripts/blob/master/sentiment_analysis_using_roberta.ipynb)|
|
||||
|[Evaluating Question Generation Models](https://github.com/flexudy-pipe/qugeev) | How accurate are the answers to questions generated by your seq2seq transformer model? | [Pascal Zoleko](https://github.com/zolekode) | [](https://colab.research.google.com/drive/1bpsSqCQU-iw_5nNoRm_crPq6FRuJthq_?usp=sharing)|
|
||||
|[Classify text with DistilBERT and Tensorflow](https://github.com/peterbayerle/huggingface_notebook/blob/main/distilbert_tf.ipynb) | How to fine-tune DistilBERT for text classification in TensorFlow | [Peter Bayerle](https://github.com/peterbayerle) | [](https://colab.research.google.com/github/peterbayerle/huggingface_notebook/blob/main/distilbert_tf.ipynb)|
|
||||
|
||||
Executable
+74
@@ -0,0 +1,74 @@
|
||||
#!/usr/bin/env python
|
||||
# coding: utf-8
|
||||
|
||||
# This script creates a super tiny model that is useful inside tests, when we just want to test that
|
||||
# the machinery works, without needing to the check the quality of the outcomes.
|
||||
#
|
||||
# This version creates a tiny vocab first, and then a tiny model - so the outcome is truly tiny -
|
||||
# all files ~60KB. As compared to taking a full-size model, reducing to the minimum its layers and
|
||||
# emb dimensions, but keeping the full vocab + merges files, leading to ~3MB in total for all files.
|
||||
# The latter is done by `fsmt-make-super-tiny-model.py`.
|
||||
#
|
||||
# It will be used then as "stas/tiny-wmt19-en-ru"
|
||||
|
||||
from pathlib import Path
|
||||
import json
|
||||
import tempfile
|
||||
|
||||
from transformers import FSMTTokenizer, FSMTConfig, FSMTForConditionalGeneration
|
||||
from transformers.tokenization_fsmt import VOCAB_FILES_NAMES
|
||||
|
||||
mname_tiny = "tiny-wmt19-en-ru"
|
||||
|
||||
# Build
|
||||
|
||||
# borrowed from a test
|
||||
vocab = [ "l", "o", "w", "e", "r", "s", "t", "i", "d", "n", "w</w>", "r</w>", "t</w>", "lo", "low", "er</w>", "low</w>", "lowest</w>", "newer</w>", "wider</w>", "<unk>", ]
|
||||
vocab_tokens = dict(zip(vocab, range(len(vocab))))
|
||||
merges = ["l o 123", "lo w 1456", "e r</w> 1789", ""]
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmpdirname:
|
||||
build_dir = Path(tmpdirname)
|
||||
src_vocab_file = build_dir / VOCAB_FILES_NAMES["src_vocab_file"]
|
||||
tgt_vocab_file = build_dir / VOCAB_FILES_NAMES["tgt_vocab_file"]
|
||||
merges_file = build_dir / VOCAB_FILES_NAMES["merges_file"]
|
||||
with open(src_vocab_file, "w") as fp: fp.write(json.dumps(vocab_tokens))
|
||||
with open(tgt_vocab_file, "w") as fp: fp.write(json.dumps(vocab_tokens))
|
||||
with open(merges_file, "w") as fp : fp.write("\n".join(merges))
|
||||
|
||||
tokenizer = FSMTTokenizer(
|
||||
langs=["en", "ru"],
|
||||
src_vocab_size = len(vocab),
|
||||
tgt_vocab_size = len(vocab),
|
||||
src_vocab_file=src_vocab_file,
|
||||
tgt_vocab_file=tgt_vocab_file,
|
||||
merges_file=merges_file,
|
||||
)
|
||||
|
||||
config = FSMTConfig(
|
||||
langs=['ru', 'en'],
|
||||
src_vocab_size=1000, tgt_vocab_size=1000,
|
||||
d_model=4,
|
||||
encoder_layers=1, decoder_layers=1,
|
||||
encoder_ffn_dim=4, decoder_ffn_dim=4,
|
||||
encoder_attention_heads=1, decoder_attention_heads=1,
|
||||
)
|
||||
|
||||
tiny_model = FSMTForConditionalGeneration(config)
|
||||
print(f"num of params {tiny_model.num_parameters()}")
|
||||
|
||||
# Test
|
||||
batch = tokenizer.prepare_seq2seq_batch(["Making tiny model"])
|
||||
outputs = tiny_model(**batch, return_dict=True)
|
||||
|
||||
print("test output:", len(outputs.logits[0]))
|
||||
|
||||
# Save
|
||||
tiny_model.half() # makes it smaller
|
||||
tiny_model.save_pretrained(mname_tiny)
|
||||
tokenizer.save_pretrained(mname_tiny)
|
||||
|
||||
print(f"Generated {mname_tiny}")
|
||||
|
||||
# Upload
|
||||
# transformers-cli upload tiny-wmt19-en-ru
|
||||
@@ -1,10 +1,19 @@
|
||||
#!/usr/bin/env python
|
||||
# coding: utf-8
|
||||
|
||||
# this script creates a tiny model that is useful inside tests, when we just want to test that the machinery works,
|
||||
# without needing to the check the quality of the outcomes.
|
||||
# it will be used then as "stas/tiny-wmt19-en-de"
|
||||
# This script creates a super tiny model that is useful inside tests, when we just want to test that
|
||||
# the machinery works, without needing to the check the quality of the outcomes.
|
||||
#
|
||||
# This version creates a tiny model through reduction of a normal pre-trained model, but keeping the
|
||||
# full vocab, merges file, and thus also resulting in a larger model due to a large vocab size.
|
||||
# This gives ~3MB in total for all files.
|
||||
#
|
||||
# If you want a 50 times smaller than this see `fsmt-make-super-tiny-model.py`, which is slightly more complicated
|
||||
#
|
||||
#
|
||||
# It will be used then as "stas/tiny-wmt19-en-de"
|
||||
|
||||
# Build
|
||||
from transformers import FSMTTokenizer, FSMTConfig, FSMTForConditionalGeneration
|
||||
mname = "facebook/wmt19-en-de"
|
||||
tokenizer = FSMTTokenizer.from_pretrained(mname)
|
||||
@@ -18,16 +27,20 @@ config.update(dict(
|
||||
|
||||
tiny_model = FSMTForConditionalGeneration(config)
|
||||
print(f"num of params {tiny_model.num_parameters()}")
|
||||
# Test it
|
||||
|
||||
# Test
|
||||
batch = tokenizer.prepare_seq2seq_batch(["Making tiny model"])
|
||||
outputs = tiny_model(**batch, return_dict=True)
|
||||
|
||||
print(len(outputs.logits[0]))
|
||||
print("test output:", len(outputs.logits[0]))
|
||||
|
||||
# Save
|
||||
mname_tiny = "tiny-wmt19-en-de"
|
||||
tiny_model.half() # makes it smaller
|
||||
tiny_model.save_pretrained(mname_tiny)
|
||||
tokenizer.save_pretrained(mname_tiny)
|
||||
|
||||
print(f"Generated {mname_tiny}")
|
||||
|
||||
# Upload
|
||||
# transformers-cli upload tiny-wmt19-en-de
|
||||
|
||||
@@ -284,6 +284,7 @@ if is_torch_available():
|
||||
DataCollatorForNextSentencePrediction,
|
||||
DataCollatorForPermutationLanguageModeling,
|
||||
DataCollatorForSOP,
|
||||
DataCollatorForWholeWordMask,
|
||||
DataCollatorWithPadding,
|
||||
default_data_collator,
|
||||
)
|
||||
@@ -291,6 +292,7 @@ if is_torch_available():
|
||||
GlueDataset,
|
||||
GlueDataTrainingArguments,
|
||||
LineByLineTextDataset,
|
||||
LineByLineWithRefDataset,
|
||||
LineByLineWithSOPTextDataset,
|
||||
SquadDataset,
|
||||
SquadDataTrainingArguments,
|
||||
@@ -652,6 +654,7 @@ if is_tf_available():
|
||||
TFAutoModelForTokenClassification,
|
||||
TFAutoModelWithLMHead,
|
||||
)
|
||||
from .modeling_tf_bart import TFBartForConditionalGeneration, TFBartModel
|
||||
from .modeling_tf_bert import (
|
||||
TF_BERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFBertEmbeddings,
|
||||
|
||||
@@ -70,6 +70,7 @@ class EncoderDecoderConfig(PretrainedConfig):
|
||||
>>> model = EncoderDecoderModel.from_pretrained('my-model', config=encoder_decoder_config)
|
||||
"""
|
||||
model_type = "encoder_decoder"
|
||||
is_composition = True
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
@@ -126,9 +126,9 @@ class FSMTConfig(PretrainedConfig):
|
||||
# update the defaults from config file
|
||||
def __init__(
|
||||
self,
|
||||
langs,
|
||||
src_vocab_size,
|
||||
tgt_vocab_size,
|
||||
langs=["en", "de"],
|
||||
src_vocab_size=42024,
|
||||
tgt_vocab_size=42024,
|
||||
activation_function="relu",
|
||||
d_model=1024,
|
||||
max_length=200,
|
||||
|
||||
@@ -77,6 +77,7 @@ RAG_CONFIG_DOC = r"""
|
||||
@add_start_docstrings(RAG_CONFIG_DOC)
|
||||
class RagConfig(PretrainedConfig):
|
||||
model_type = "rag"
|
||||
is_composition = True
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
|
||||
@@ -41,6 +41,10 @@ class PretrainedConfig(object):
|
||||
Class attributes (overridden by derived classes)
|
||||
- **model_type** (:obj:`str`): An identifier for the model type, serialized into the JSON file, and used to
|
||||
recreate the correct object in :class:`~transformers.AutoConfig`.
|
||||
- **is_composition** (:obj:`bool`): Whether the config class is composed of multiple
|
||||
sub-configs. In this case the config has to be initialized from two or more configs of
|
||||
type :class:`~transformers.PretrainedConfig` like: :class:`~transformers.EncoderDecoderConfig` or
|
||||
:class:`~RagConfig`.
|
||||
|
||||
Args:
|
||||
name_or_path (:obj:`str`, `optional`, defaults to :obj:`""`):
|
||||
@@ -145,6 +149,7 @@ class PretrainedConfig(object):
|
||||
use BFloat16 scalars (only used by some TensorFlow models).
|
||||
"""
|
||||
model_type: str = ""
|
||||
is_composition: bool = False
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
# Attributes with defaults
|
||||
@@ -476,11 +481,18 @@ class PretrainedConfig(object):
|
||||
# get the default config dict
|
||||
default_config_dict = PretrainedConfig().to_dict()
|
||||
|
||||
# get class specific config dict
|
||||
class_config_dict = self.__class__().to_dict() if not self.is_composition else {}
|
||||
|
||||
serializable_config_dict = {}
|
||||
|
||||
# only serialize values that differ from the default config
|
||||
for key, value in config_dict.items():
|
||||
if key not in default_config_dict or value != default_config_dict[key]:
|
||||
if (
|
||||
key not in default_config_dict
|
||||
or value != default_config_dict[key]
|
||||
or (key in class_config_dict and value != class_config_dict[key])
|
||||
):
|
||||
serializable_config_dict[key] = value
|
||||
|
||||
return serializable_config_dict
|
||||
|
||||
@@ -20,6 +20,7 @@ import os
|
||||
|
||||
from transformers import (
|
||||
ALBERT_PRETRAINED_CONFIG_ARCHIVE_MAP,
|
||||
BART_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
BERT_PRETRAINED_CONFIG_ARCHIVE_MAP,
|
||||
CAMEMBERT_PRETRAINED_CONFIG_ARCHIVE_MAP,
|
||||
CTRL_PRETRAINED_CONFIG_ARCHIVE_MAP,
|
||||
@@ -37,6 +38,7 @@ from transformers import (
|
||||
XLM_ROBERTA_PRETRAINED_CONFIG_ARCHIVE_MAP,
|
||||
XLNET_PRETRAINED_CONFIG_ARCHIVE_MAP,
|
||||
AlbertConfig,
|
||||
BartConfig,
|
||||
BertConfig,
|
||||
CamembertConfig,
|
||||
CTRLConfig,
|
||||
@@ -49,6 +51,7 @@ from transformers import (
|
||||
RobertaConfig,
|
||||
T5Config,
|
||||
TFAlbertForPreTraining,
|
||||
TFBartForConditionalGeneration,
|
||||
TFBertForPreTraining,
|
||||
TFBertForQuestionAnswering,
|
||||
TFBertForSequenceClassification,
|
||||
@@ -87,6 +90,7 @@ if is_torch_available():
|
||||
|
||||
from transformers import (
|
||||
AlbertForPreTraining,
|
||||
BartForConditionalGeneration,
|
||||
BertForPreTraining,
|
||||
BertForQuestionAnswering,
|
||||
BertForSequenceClassification,
|
||||
@@ -113,6 +117,12 @@ if is_torch_available():
|
||||
logging.set_verbosity_info()
|
||||
|
||||
MODEL_CLASSES = {
|
||||
"bart": (
|
||||
BartConfig,
|
||||
TFBartForConditionalGeneration,
|
||||
BartForConditionalGeneration,
|
||||
BART_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
),
|
||||
"bert": (
|
||||
BertConfig,
|
||||
TFBertForPreTraining,
|
||||
|
||||
@@ -78,15 +78,16 @@ def convert_slow_checkpoint_to_fast(tokenizer_name, checkpoint_name, dump_path,
|
||||
"=> {} with prefix {}, add_prefix {}".format(dump_path_full, checkpoint_prefix_name, add_prefix)
|
||||
)
|
||||
|
||||
file_path = list(tokenizer.pretrained_vocab_files_map.values())[0][checkpoint]
|
||||
next_char = file_path.split(checkpoint)[-1][0]
|
||||
if next_char == "/":
|
||||
dump_path_full = os.path.join(dump_path_full, checkpoint_prefix_name)
|
||||
checkpoint_prefix_name = None
|
||||
if checkpoint in list(tokenizer.pretrained_vocab_files_map.values())[0]:
|
||||
file_path = list(tokenizer.pretrained_vocab_files_map.values())[0][checkpoint]
|
||||
next_char = file_path.split(checkpoint)[-1][0]
|
||||
if next_char == "/":
|
||||
dump_path_full = os.path.join(dump_path_full, checkpoint_prefix_name)
|
||||
checkpoint_prefix_name = None
|
||||
|
||||
logger.info(
|
||||
"=> {} with prefix {}, add_prefix {}".format(dump_path_full, checkpoint_prefix_name, add_prefix)
|
||||
)
|
||||
logger.info(
|
||||
"=> {} with prefix {}, add_prefix {}".format(dump_path_full, checkpoint_prefix_name, add_prefix)
|
||||
)
|
||||
|
||||
file_names = tokenizer.save_pretrained(
|
||||
dump_path_full, legacy_format=False, filename_prefix=checkpoint_prefix_name
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import random
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable, Dict, List, NewType, Optional, Tuple, Union
|
||||
|
||||
@@ -195,6 +196,124 @@ class DataCollatorForLanguageModeling:
|
||||
return inputs, labels
|
||||
|
||||
|
||||
@dataclass
|
||||
class DataCollatorForWholeWordMask(DataCollatorForLanguageModeling):
|
||||
"""
|
||||
Data collator used for language modeling.
|
||||
- collates batches of tensors, honoring their tokenizer's pad_token
|
||||
- preprocesses batches for masked language modeling
|
||||
"""
|
||||
|
||||
def __call__(
|
||||
self, examples: List[Union[List[int], torch.Tensor, Dict[str, torch.Tensor]]]
|
||||
) -> Dict[str, torch.Tensor]:
|
||||
if isinstance(examples[0], (dict, BatchEncoding)):
|
||||
input_ids = [e["input_ids"] for e in examples]
|
||||
else:
|
||||
input_ids = examples
|
||||
examples = [{"input_ids": e} for e in examples]
|
||||
|
||||
batch_input = self._tensorize_batch(input_ids)
|
||||
|
||||
mask_labels = []
|
||||
for e in examples:
|
||||
ref_tokens = []
|
||||
for id in e["input_ids"].tolist():
|
||||
token = self.tokenizer._convert_id_to_token(id)
|
||||
ref_tokens.append(token)
|
||||
|
||||
# For Chinese tokens, we need extra inf to mark sub-word, e.g [喜,欢]-> [喜,##欢]
|
||||
if "chinese_ref" in e:
|
||||
ref_pos = e["chinese_ref"].tolist()
|
||||
len_seq = e["input_ids"].size(0)
|
||||
for i in range(len_seq):
|
||||
if i in ref_pos:
|
||||
ref_tokens[i] = "##" + ref_tokens[i]
|
||||
mask_labels.append(self._whole_word_mask(ref_tokens))
|
||||
batch_mask = self._tensorize_batch(mask_labels)
|
||||
inputs, labels = self.mask_tokens(batch_input, batch_mask)
|
||||
return {"input_ids": inputs, "labels": labels}
|
||||
|
||||
def _whole_word_mask(self, input_tokens: List[str], max_predictions=512):
|
||||
"""
|
||||
Get 0/1 labels for masked tokens with whole word mask proxy
|
||||
"""
|
||||
|
||||
cand_indexes = []
|
||||
for (i, token) in enumerate(input_tokens):
|
||||
if token == "[CLS]" or token == "[SEP]":
|
||||
continue
|
||||
|
||||
if len(cand_indexes) >= 1 and token.startswith("##"):
|
||||
cand_indexes[-1].append(i)
|
||||
else:
|
||||
cand_indexes.append([i])
|
||||
|
||||
random.shuffle(cand_indexes)
|
||||
num_to_predict = min(max_predictions, max(1, int(round(len(input_tokens) * self.mlm_probability))))
|
||||
masked_lms = []
|
||||
covered_indexes = set()
|
||||
for index_set in cand_indexes:
|
||||
if len(masked_lms) >= num_to_predict:
|
||||
break
|
||||
# If adding a whole-word mask would exceed the maximum number of
|
||||
# predictions, then just skip this candidate.
|
||||
if len(masked_lms) + len(index_set) > num_to_predict:
|
||||
continue
|
||||
is_any_index_covered = False
|
||||
for index in index_set:
|
||||
if index in covered_indexes:
|
||||
is_any_index_covered = True
|
||||
break
|
||||
if is_any_index_covered:
|
||||
continue
|
||||
for index in index_set:
|
||||
covered_indexes.add(index)
|
||||
masked_lms.append(index)
|
||||
|
||||
assert len(covered_indexes) == len(masked_lms)
|
||||
mask_labels = [1 if i in covered_indexes else 0 for i in range(len(input_tokens))]
|
||||
return mask_labels
|
||||
|
||||
def mask_tokens(self, inputs: torch.Tensor, mask_labels: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
"""
|
||||
Prepare masked tokens inputs/labels for masked language modeling: 80% MASK, 10% random, 10% original.
|
||||
Set 'mask_labels' means we use whole word mask (wwm), we directly mask idxs according to it's ref.
|
||||
"""
|
||||
|
||||
if self.tokenizer.mask_token is None:
|
||||
raise ValueError(
|
||||
"This tokenizer does not have a mask token which is necessary for masked language modeling. Remove the --mlm flag if you want to use this tokenizer."
|
||||
)
|
||||
labels = inputs.clone()
|
||||
# We sample a few tokens in each sequence for masked-LM training (with probability args.mlm_probability defaults to 0.15 in Bert/RoBERTa)
|
||||
|
||||
probability_matrix = mask_labels
|
||||
|
||||
special_tokens_mask = [
|
||||
self.tokenizer.get_special_tokens_mask(val, already_has_special_tokens=True) for val in labels.tolist()
|
||||
]
|
||||
probability_matrix.masked_fill_(torch.tensor(special_tokens_mask, dtype=torch.bool), value=0.0)
|
||||
if self.tokenizer._pad_token is not None:
|
||||
padding_mask = labels.eq(self.tokenizer.pad_token_id)
|
||||
probability_matrix.masked_fill_(padding_mask, value=0.0)
|
||||
|
||||
masked_indices = probability_matrix.bool()
|
||||
labels[~masked_indices] = -100 # We only compute loss on masked tokens
|
||||
|
||||
# 80% of the time, we replace masked input tokens with tokenizer.mask_token ([MASK])
|
||||
indices_replaced = torch.bernoulli(torch.full(labels.shape, 0.8)).bool() & masked_indices
|
||||
inputs[indices_replaced] = self.tokenizer.convert_tokens_to_ids(self.tokenizer.mask_token)
|
||||
|
||||
# 10% of the time, we replace masked input tokens with random word
|
||||
indices_random = torch.bernoulli(torch.full(labels.shape, 0.5)).bool() & masked_indices & ~indices_replaced
|
||||
random_words = torch.randint(len(self.tokenizer), labels.shape, dtype=torch.long)
|
||||
inputs[indices_random] = random_words[indices_random]
|
||||
|
||||
# The rest of the time (10% of the time) we keep the masked input tokens unchanged
|
||||
return inputs, labels
|
||||
|
||||
|
||||
@dataclass
|
||||
class DataCollatorForSOP(DataCollatorForLanguageModeling):
|
||||
"""
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
from .glue import GlueDataset, GlueDataTrainingArguments
|
||||
from .language_modeling import (
|
||||
LineByLineTextDataset,
|
||||
LineByLineWithRefDataset,
|
||||
LineByLineWithSOPTextDataset,
|
||||
TextDataset,
|
||||
TextDatasetForNextSentencePrediction,
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import json
|
||||
import os
|
||||
import pickle
|
||||
import random
|
||||
@@ -106,12 +107,48 @@ class LineByLineTextDataset(Dataset):
|
||||
|
||||
batch_encoding = tokenizer(lines, add_special_tokens=True, truncation=True, max_length=block_size)
|
||||
self.examples = batch_encoding["input_ids"]
|
||||
self.examples = [{"input_ids": torch.tensor(e, dtype=torch.long)} for e in self.examples]
|
||||
|
||||
def __len__(self):
|
||||
return len(self.examples)
|
||||
|
||||
def __getitem__(self, i) -> torch.Tensor:
|
||||
return torch.tensor(self.examples[i], dtype=torch.long)
|
||||
def __getitem__(self, i) -> Dict[str, torch.tensor]:
|
||||
return self.examples[i]
|
||||
|
||||
|
||||
class LineByLineWithRefDataset(Dataset):
|
||||
"""
|
||||
This will be superseded by a framework-agnostic approach
|
||||
soon.
|
||||
"""
|
||||
|
||||
def __init__(self, tokenizer: PreTrainedTokenizer, file_path: str, block_size: int, ref_path: str):
|
||||
assert os.path.isfile(file_path), f"Input file path {file_path} not found"
|
||||
assert os.path.isfile(ref_path), f"Ref file path {file_path} not found"
|
||||
# Here, we do not cache the features, operating under the assumption
|
||||
# that we will soon use fast multithreaded tokenizers from the
|
||||
# `tokenizers` repo everywhere =)
|
||||
logger.info("Creating features from dataset file at %s", file_path)
|
||||
logger.info("Use ref segment results at %s", ref_path)
|
||||
with open(file_path, encoding="utf-8") as f:
|
||||
data = [line for line in f.read().splitlines() if (len(line) > 0 and not line.isspace())]
|
||||
batch_encoding = tokenizer(data, add_special_tokens=True, truncation=True, max_length=block_size)
|
||||
self.examples = batch_encoding["input_ids"]
|
||||
self.examples = [{"input_ids": torch.tensor(e, dtype=torch.long)} for e in self.examples]
|
||||
|
||||
# Get ref inf from file
|
||||
with open(ref_path, encoding="utf-8") as f:
|
||||
ref = [json.loads(line) for line in f.read().splitlines() if (len(line) > 0 and not line.isspace())]
|
||||
assert len(data) == len(ref)
|
||||
n = len(self.examples)
|
||||
for i in range(n):
|
||||
self.examples[i]["chinese_ref"] = torch.tensor(ref[i], dtype=torch.long)
|
||||
|
||||
def __len__(self):
|
||||
return len(self.examples)
|
||||
|
||||
def __getitem__(self, i) -> Dict[str, torch.tensor]:
|
||||
return self.examples[i]
|
||||
|
||||
|
||||
class LineByLineWithSOPTextDataset(Dataset):
|
||||
|
||||
@@ -7,11 +7,8 @@ import numpy as np
|
||||
from tqdm import tqdm
|
||||
|
||||
from ...file_utils import is_tf_available, is_torch_available
|
||||
from ...tokenization_bart import BartTokenizer
|
||||
from ...tokenization_bert import whitespace_tokenize
|
||||
from ...tokenization_longformer import LongformerTokenizer
|
||||
from ...tokenization_roberta import RobertaTokenizer
|
||||
from ...tokenization_utils_base import TruncationStrategy
|
||||
from ...tokenization_utils_base import PreTrainedTokenizerBase, TruncationStrategy
|
||||
from ...utils import logging
|
||||
from .utils import DataProcessor
|
||||
|
||||
@@ -112,7 +109,14 @@ def squad_convert_example_to_features(
|
||||
all_doc_tokens = []
|
||||
for (i, token) in enumerate(example.doc_tokens):
|
||||
orig_to_tok_index.append(len(all_doc_tokens))
|
||||
if isinstance(tokenizer, (RobertaTokenizer, LongformerTokenizer, BartTokenizer)):
|
||||
if tokenizer.__class__.__name__ in [
|
||||
"RobertaTokenizer",
|
||||
"LongformerTokenizer",
|
||||
"BartTokenizer",
|
||||
"RobertaTokenizerFast",
|
||||
"LongformerTokenizerFast",
|
||||
"BartTokenizerFast",
|
||||
]:
|
||||
sub_tokens = tokenizer.tokenize(token, add_prefix_space=True)
|
||||
else:
|
||||
sub_tokens = tokenizer.tokenize(token)
|
||||
@@ -292,7 +296,7 @@ def squad_convert_example_to_features(
|
||||
return features
|
||||
|
||||
|
||||
def squad_convert_example_to_features_init(tokenizer_for_convert):
|
||||
def squad_convert_example_to_features_init(tokenizer_for_convert: PreTrainedTokenizerBase):
|
||||
global tokenizer
|
||||
tokenizer = tokenizer_for_convert
|
||||
|
||||
@@ -344,9 +348,9 @@ def squad_convert_examples_to_features(
|
||||
is_training=not evaluate,
|
||||
)
|
||||
"""
|
||||
|
||||
# Defining helper methods
|
||||
features = []
|
||||
|
||||
threads = min(threads, cpu_count())
|
||||
with Pool(threads, initializer=squad_convert_example_to_features_init, initargs=(tokenizer,)) as p:
|
||||
annotate_ = partial(
|
||||
@@ -365,6 +369,7 @@ def squad_convert_examples_to_features(
|
||||
disable=not tqdm_enabled,
|
||||
)
|
||||
)
|
||||
|
||||
new_features = []
|
||||
unique_id = 1000000000
|
||||
example_index = 0
|
||||
|
||||
@@ -640,6 +640,10 @@ class TFGenerationMixin:
|
||||
if temperature != 1.0:
|
||||
next_token_logits = next_token_logits / temperature
|
||||
|
||||
if self.config.is_encoder_decoder and do_sample is False:
|
||||
next_token_logits = self.adjust_logits_during_generation(
|
||||
next_token_logits, cur_len=cur_len, max_length=max_length
|
||||
)
|
||||
# calculate log softmax score
|
||||
scores = tf.nn.log_softmax(next_token_logits, axis=-1) # (batch_size * num_beams, vocab_size)
|
||||
|
||||
@@ -890,6 +894,13 @@ class TFGenerationMixin:
|
||||
def _reorder_cache(past, beam_idx):
|
||||
return tuple(tf.gather(layer_past, beam_idx, axis=1) for layer_past in past)
|
||||
|
||||
def adjust_logits_during_generation(self, logits, **kwargs):
|
||||
"""
|
||||
Implement in subclasses of :class:`~transfomers.PreTrainedModel` for custom behavior to adjust the logits in
|
||||
the generate method.
|
||||
"""
|
||||
return logits
|
||||
|
||||
|
||||
def _create_next_token_logits_penalties(input_ids, logits, repetition_penalty):
|
||||
# create logit penalties for already seen input_ids
|
||||
|
||||
@@ -85,6 +85,17 @@ def is_ray_available():
|
||||
return _has_ray
|
||||
|
||||
|
||||
def hp_params(trial):
|
||||
if is_optuna_available():
|
||||
if isinstance(trial, optuna.Trial):
|
||||
return trial.params
|
||||
if is_ray_available():
|
||||
if isinstance(trial, dict):
|
||||
return trial
|
||||
|
||||
raise RuntimeError(f"Unknown type for trial {trial.__class__}")
|
||||
|
||||
|
||||
def default_hp_search_backend():
|
||||
if is_optuna_available():
|
||||
return "optuna"
|
||||
@@ -192,6 +203,18 @@ def run_hp_search_ray(trainer, n_trials: int, direction: str, **kwargs) -> BestR
|
||||
return best_run
|
||||
|
||||
|
||||
def rewrite_logs(d):
|
||||
new_d = {}
|
||||
eval_prefix = "eval_"
|
||||
eval_prefix_len = len(eval_prefix)
|
||||
for k, v in d.items():
|
||||
if k.startswith(eval_prefix):
|
||||
new_d["eval/" + k[eval_prefix_len:]] = v
|
||||
else:
|
||||
new_d["train/" + k] = v
|
||||
return new_d
|
||||
|
||||
|
||||
class TensorBoardCallback(TrainerCallback):
|
||||
"""
|
||||
A :class:`~transformers.TrainerCallback` that sends the logs to `TensorBoard
|
||||
@@ -208,17 +231,39 @@ class TensorBoardCallback(TrainerCallback):
|
||||
), "TensorBoardCallback requires tensorboard to be installed. Either update your PyTorch version or install tensorboardX."
|
||||
self.tb_writer = tb_writer
|
||||
|
||||
def on_init_end(self, args, state, control, **kwargs):
|
||||
if self.tb_writer is None and state.is_world_process_zero:
|
||||
self.tb_writer = SummaryWriter(log_dir=args.logging_dir)
|
||||
def _init_summary_writer(self, args, log_dir=None):
|
||||
log_dir = log_dir or args.logging_dir
|
||||
self.tb_writer = SummaryWriter(log_dir=log_dir)
|
||||
|
||||
def on_train_begin(self, args, state, control, **kwargs):
|
||||
if not state.is_world_process_zero:
|
||||
return
|
||||
|
||||
log_dir = None
|
||||
|
||||
if state.is_hyper_param_search:
|
||||
trial_name = state.trial_name
|
||||
if trial_name is not None:
|
||||
log_dir = os.path.join(args.logging_dir, trial_name)
|
||||
|
||||
self._init_summary_writer(args, log_dir)
|
||||
|
||||
if self.tb_writer is not None:
|
||||
self.tb_writer.add_text("args", args.to_json_string())
|
||||
if "model" in kwargs:
|
||||
model = kwargs["model"]
|
||||
if hasattr(model, "config") and model.config is not None:
|
||||
model_config_json = model.config.to_json_string()
|
||||
self.tb_writer.add_text("model_config", model_config_json)
|
||||
self.tb_writer.add_hparams(args.to_sanitized_dict(), metric_dict={})
|
||||
|
||||
def on_log(self, args, state, control, logs=None, **kwargs):
|
||||
if state.is_world_process_zero:
|
||||
if self.tb_writer is None:
|
||||
self._init_summary_writer(args)
|
||||
|
||||
if self.tb_writer:
|
||||
logs = rewrite_logs(logs)
|
||||
for k, v in logs.items():
|
||||
if isinstance(v, (int, float)):
|
||||
self.tb_writer.add_scalar(k, v, state.global_step)
|
||||
@@ -249,7 +294,7 @@ class WandbCallback(TrainerCallback):
|
||||
assert _has_wandb, "WandbCallback requires wandb to be installed. Run `pip install wandb`."
|
||||
self._initialized = False
|
||||
|
||||
def setup(self, args, state, model):
|
||||
def setup(self, args, state, model, reinit, **kwargs):
|
||||
"""
|
||||
Setup the optional Weights & Biases (`wandb`) integration.
|
||||
|
||||
@@ -271,21 +316,41 @@ class WandbCallback(TrainerCallback):
|
||||
'Automatic Weights & Biases logging enabled, to disable set os.environ["WANDB_DISABLED"] = "true"'
|
||||
)
|
||||
combined_dict = {**args.to_sanitized_dict()}
|
||||
if getattr(model, "config", None) is not None:
|
||||
combined_dict = {**model.config.to_dict(), **combined_dict}
|
||||
wandb.init(project=os.getenv("WANDB_PROJECT", "huggingface"), config=combined_dict, name=args.run_name)
|
||||
|
||||
if hasattr(model, "config") and model.config is not None:
|
||||
model_config = model.config.to_dict()
|
||||
combined_dict = {**model_config, **combined_dict}
|
||||
trial_name = state.trial_name
|
||||
init_args = {}
|
||||
if trial_name is not None:
|
||||
run_name = trial_name
|
||||
init_args["group"] = args.run_name
|
||||
else:
|
||||
run_name = args.run_name
|
||||
|
||||
wandb.init(
|
||||
project=os.getenv("WANDB_PROJECT", "huggingface"),
|
||||
config=combined_dict,
|
||||
name=run_name,
|
||||
reinit=reinit,
|
||||
**init_args,
|
||||
)
|
||||
|
||||
# keep track of model topology and gradients, unsupported on TPU
|
||||
if not is_torch_tpu_available() and os.getenv("WANDB_WATCH") != "false":
|
||||
wandb.watch(model, log=os.getenv("WANDB_WATCH", "gradients"), log_freq=max(100, args.logging_steps))
|
||||
|
||||
def on_train_begin(self, args, state, control, model=None, **kwargs):
|
||||
if not self._initialized:
|
||||
self.setup(args, state, model)
|
||||
hp_search = state.is_hyper_param_search
|
||||
if not self._initialized or hp_search:
|
||||
print(args.run_name)
|
||||
self.setup(args, state, model, reinit=hp_search, **kwargs)
|
||||
|
||||
def on_log(self, args, state, control, model=None, logs=None, **kwargs):
|
||||
if not self._initialized:
|
||||
self.setup(args, state, model)
|
||||
self.setup(args, state, model, reinit=False)
|
||||
if state.is_world_process_zero:
|
||||
logs = rewrite_logs(logs)
|
||||
wandb.log(logs, step=state.global_step)
|
||||
|
||||
|
||||
|
||||
@@ -335,7 +335,7 @@ MODEL_FOR_CAUSAL_LM_MAPPING = OrderedDict(
|
||||
(CTRLConfig, CTRLLMHeadModel),
|
||||
(ReformerConfig, ReformerModelWithLMHead),
|
||||
(BertGenerationConfig, BertGenerationDecoder),
|
||||
(ProphetNetConfig, XLMProphetNetForCausalLM),
|
||||
(XLMProphetNetConfig, XLMProphetNetForCausalLM),
|
||||
(ProphetNetConfig, ProphetNetForCausalLM),
|
||||
]
|
||||
)
|
||||
|
||||
@@ -131,7 +131,7 @@ BART_INPUTS_DOCSTRING = r"""
|
||||
:obj:`last_hidden_state` of shape :obj:`(batch_size, sequence_length, hidden_size)`, `optional`) is a
|
||||
sequence of hidden-states at the output of the last layer of the encoder. Used in the cross-attention of
|
||||
the decoder.
|
||||
past_key_values (:obj:`tuple(tuple(torch.FloatTensor))` of length :obj:`config.n_layers` with each tuple having 4 tensors of shape :obj:`(batch_size, num_heads, sequence_length - 1, embed_size_per_head)`):
|
||||
past_key_values (:obj:`Tuple[Dict[str: tf.Tensor]]` of length :obj:`config.n_layers` with each tuple having 4 tensors of shape :obj:`(batch_size, num_heads, sequence_length - 1, embed_size_per_head)`):
|
||||
Contains precomputed key and value hidden-states of the attention blocks. Can be used to speed up decoding.
|
||||
|
||||
If :obj:`past_key_values` are used, the user can optionally input only the last
|
||||
@@ -217,12 +217,6 @@ def _make_linear_from_emb(emb):
|
||||
return lin_layer
|
||||
|
||||
|
||||
# Helper Functions, mostly for making masks
|
||||
def _check_shapes(shape_1, shape2):
|
||||
if shape_1 != shape2:
|
||||
raise AssertionError("shape mismatch: {} != {}".format(shape_1, shape2))
|
||||
|
||||
|
||||
def shift_tokens_right(input_ids, pad_token_id):
|
||||
"""Shift input ids one token to the right, and wrap the last non pad token (usually <eos>)."""
|
||||
prev_output_tokens = input_ids.clone()
|
||||
@@ -595,7 +589,7 @@ class BartDecoder(nn.Module):
|
||||
# decoder layers
|
||||
all_hidden_states = () if output_hidden_states else None
|
||||
all_self_attns = () if output_attentions else None
|
||||
next_decoder_cache = []
|
||||
next_decoder_cache: List[Dict] = []
|
||||
for idx, decoder_layer in enumerate(self.layers):
|
||||
# add LayerDrop (see https://arxiv.org/abs/1909.11556 for description)
|
||||
if output_hidden_states:
|
||||
@@ -640,7 +634,7 @@ class BartDecoder(nn.Module):
|
||||
)
|
||||
|
||||
|
||||
def _reorder_buffer(attn_cache, new_order):
|
||||
def _reorder_buffer(attn_cache: Dict, new_order) -> Dict:
|
||||
for k, input_buffer_k in attn_cache.items():
|
||||
if input_buffer_k is not None:
|
||||
attn_cache[k] = input_buffer_k.index_select(0, new_order)
|
||||
@@ -679,17 +673,15 @@ class Attention(nn.Module):
|
||||
def forward(
|
||||
self,
|
||||
query,
|
||||
key: Optional[Tensor],
|
||||
key: Tensor,
|
||||
key_padding_mask: Optional[Tensor] = None,
|
||||
layer_state: Optional[Dict[str, Optional[Tensor]]] = None,
|
||||
layer_state: Optional[Dict[str, Tensor]] = None,
|
||||
attn_mask: Optional[Tensor] = None,
|
||||
output_attentions=False,
|
||||
) -> Tuple[Tensor, Optional[Tensor]]:
|
||||
"""Input shape: Time(SeqLen) x Batch x Channel"""
|
||||
static_kv: bool = self.encoder_decoder_attention
|
||||
tgt_len, bsz, embed_dim = query.size()
|
||||
assert embed_dim == self.embed_dim
|
||||
assert list(query.size()) == [tgt_len, bsz, embed_dim]
|
||||
# get here for encoder decoder cause of static_kv
|
||||
if layer_state is not None: # reuse k,v and encoder_padding_mask
|
||||
saved_state = layer_state.get(self.cache_key, {})
|
||||
@@ -697,17 +689,16 @@ class Attention(nn.Module):
|
||||
# previous time steps are cached - no need to recompute key and value if they are static
|
||||
key = None
|
||||
else:
|
||||
# this branch is hit by encoder
|
||||
saved_state = None
|
||||
layer_state = {}
|
||||
|
||||
q = self.q_proj(query) * self.scaling
|
||||
if static_kv:
|
||||
if key is None:
|
||||
k = v = None
|
||||
else:
|
||||
k = self.k_proj(key)
|
||||
v = self.v_proj(key)
|
||||
else:
|
||||
if static_kv and key is None: # cross-attention with cache
|
||||
k = v = None
|
||||
elif static_kv and key is not None: # cross-attention no prev_key found in cache
|
||||
k = self.k_proj(key)
|
||||
v = self.v_proj(key)
|
||||
else: # self-attention
|
||||
k = self.k_proj(query)
|
||||
v = self.v_proj(query)
|
||||
|
||||
@@ -717,18 +708,16 @@ class Attention(nn.Module):
|
||||
if v is not None:
|
||||
v = self._shape(v, -1, bsz)
|
||||
|
||||
if saved_state is not None:
|
||||
k, v, key_padding_mask = self._use_saved_state(k, v, saved_state, key_padding_mask, static_kv, bsz)
|
||||
if saved_state:
|
||||
k, v = self._concat_saved_state(k, v, saved_state, static_kv, bsz)
|
||||
|
||||
# Update cache
|
||||
layer_state[self.cache_key] = {
|
||||
"prev_key": k.view(bsz, self.num_heads, -1, self.head_dim),
|
||||
"prev_value": v.view(bsz, self.num_heads, -1, self.head_dim),
|
||||
"prev_key_padding_mask": key_padding_mask if not static_kv else None,
|
||||
}
|
||||
if isinstance(layer_state, dict):
|
||||
cached_shape = (bsz, self.num_heads, -1, self.head_dim) # bsz must be first for reorder_cache
|
||||
layer_state[self.cache_key] = dict(prev_key=k.view(*cached_shape), prev_value=v.view(*cached_shape))
|
||||
|
||||
assert k is not None
|
||||
src_len = k.size(1)
|
||||
assert key_padding_mask is None or key_padding_mask.shape == (bsz, src_len)
|
||||
attn_weights = torch.bmm(q, k.transpose(1, 2))
|
||||
assert attn_weights.size() == (bsz * self.num_heads, tgt_len, src_len)
|
||||
|
||||
@@ -736,13 +725,7 @@ class Attention(nn.Module):
|
||||
attn_weights = attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attn_mask
|
||||
attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len)
|
||||
|
||||
# This is part of a workaround to get around fork/join parallelism not supporting Optional types.
|
||||
if key_padding_mask is not None and key_padding_mask.dim() == 0:
|
||||
key_padding_mask = None
|
||||
assert key_padding_mask is None or key_padding_mask.size()[:2] == (
|
||||
bsz,
|
||||
src_len,
|
||||
)
|
||||
# Note: deleted workaround to get around fork/join parallelism not supporting Optional types. on 2020/10/15
|
||||
|
||||
if key_padding_mask is not None: # don't attend to padding symbols
|
||||
attn_weights = attn_weights.view(bsz, self.num_heads, tgt_len, src_len)
|
||||
@@ -750,11 +733,7 @@ class Attention(nn.Module):
|
||||
attn_weights = attn_weights.masked_fill(reshaped, float("-inf"))
|
||||
attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len)
|
||||
attn_weights = F.softmax(attn_weights, dim=-1)
|
||||
attn_probs = F.dropout(
|
||||
attn_weights,
|
||||
p=self.dropout,
|
||||
training=self.training,
|
||||
)
|
||||
attn_probs = F.dropout(attn_weights, p=self.dropout, training=self.training)
|
||||
|
||||
assert v is not None
|
||||
attn_output = torch.bmm(attn_probs, v)
|
||||
@@ -767,36 +746,13 @@ class Attention(nn.Module):
|
||||
attn_weights = None
|
||||
return attn_output, attn_weights
|
||||
|
||||
def _use_saved_state(self, k, v, saved_state, key_padding_mask, static_kv, bsz):
|
||||
def _concat_saved_state(self, k, v, saved_state, static_kv, bsz) -> Tuple[Tensor]:
|
||||
# saved states are stored with shape (bsz, num_heads, seq_len, head_dim)
|
||||
if "prev_key" in saved_state:
|
||||
_prev_key = saved_state["prev_key"]
|
||||
assert _prev_key is not None
|
||||
prev_key = _prev_key.view(bsz * self.num_heads, -1, self.head_dim)
|
||||
if static_kv:
|
||||
k = prev_key
|
||||
else:
|
||||
assert k is not None
|
||||
k = torch.cat([prev_key, k], dim=1)
|
||||
if "prev_value" in saved_state:
|
||||
_prev_value = saved_state["prev_value"]
|
||||
assert _prev_value is not None
|
||||
prev_value = _prev_value.view(bsz * self.num_heads, -1, self.head_dim)
|
||||
if static_kv:
|
||||
v = prev_value
|
||||
else:
|
||||
assert v is not None
|
||||
v = torch.cat([prev_value, v], dim=1)
|
||||
assert k is not None and v is not None
|
||||
prev_key_padding_mask: Optional[Tensor] = saved_state.get("prev_key_padding_mask", None)
|
||||
if prev_key_padding_mask is not None:
|
||||
if static_kv:
|
||||
new_key_padding_mask = prev_key_padding_mask
|
||||
else:
|
||||
new_key_padding_mask = torch.cat([prev_key_padding_mask, key_padding_mask], dim=1)
|
||||
else:
|
||||
new_key_padding_mask = key_padding_mask
|
||||
return k, v, new_key_padding_mask
|
||||
prev_K = saved_state["prev_key"].view(bsz * self.num_heads, -1, self.head_dim)
|
||||
prev_V = saved_state["prev_value"].view(bsz * self.num_heads, -1, self.head_dim)
|
||||
new_K = prev_K if static_kv else torch.cat([prev_K, k], dim=1)
|
||||
new_V = prev_V if static_kv else torch.cat([prev_V, v], dim=1)
|
||||
return new_K, new_V
|
||||
|
||||
|
||||
class BartClassificationHead(nn.Module):
|
||||
@@ -1143,14 +1099,15 @@ class BartForConditionalGeneration(PretrainedBartModel):
|
||||
|
||||
def adjust_logits_during_generation(self, logits, cur_len, max_length):
|
||||
if cur_len == 1 and self.config.force_bos_token_to_be_generated:
|
||||
self._force_token_ids_generation(logits, self.config.bos_token_id)
|
||||
self._force_token_id_to_be_generated(logits, self.config.bos_token_id)
|
||||
elif cur_len == max_length - 1 and self.config.eos_token_id is not None:
|
||||
self._force_token_ids_generation(logits, self.config.eos_token_id)
|
||||
self._force_token_id_to_be_generated(logits, self.config.eos_token_id)
|
||||
return logits
|
||||
|
||||
def _force_token_ids_generation(self, scores, token_id) -> None:
|
||||
@staticmethod
|
||||
def _force_token_id_to_be_generated(scores, token_id) -> None:
|
||||
"""force one of token_ids to be generated by setting prob of all other tokens to 0 (logprob=-float("inf"))"""
|
||||
scores[:, [x for x in range(self.config.vocab_size) if x != token_id]] = -float("inf")
|
||||
scores[:, [x for x in range(scores.shape[1]) if x != token_id]] = -float("inf")
|
||||
|
||||
@staticmethod
|
||||
def _reorder_cache(past, beam_idx):
|
||||
|
||||
@@ -52,5 +52,5 @@ class BlenderbotForConditionalGeneration(BartForConditionalGeneration):
|
||||
def adjust_logits_during_generation(self, logits, cur_len, max_length):
|
||||
logits[:, self.config.bos_token_id] = -torch.finfo(torch.float16).max # near infinity fp16
|
||||
if cur_len == max_length - 1 and self.config.eos_token_id is not None:
|
||||
self._force_token_ids_generation(logits, self.config.eos_token_id)
|
||||
self._force_token_id_to_be_generated(logits, self.config.eos_token_id)
|
||||
return logits
|
||||
|
||||
@@ -607,11 +607,12 @@ class GPT2Model(GPT2PreTrainedModel):
|
||||
if inputs_embeds is None:
|
||||
inputs_embeds = self.wte(input_ids)
|
||||
position_embeds = self.wpe(position_ids)
|
||||
hidden_states = inputs_embeds + position_embeds
|
||||
|
||||
if token_type_ids is not None:
|
||||
token_type_embeds = self.wte(token_type_ids)
|
||||
else:
|
||||
token_type_embeds = 0
|
||||
hidden_states = inputs_embeds + position_embeds + token_type_embeds
|
||||
hidden_states = hidden_states + token_type_embeds
|
||||
|
||||
hidden_states = self.drop(hidden_states)
|
||||
|
||||
output_shape = input_shape + (hidden_states.size(-1),)
|
||||
|
||||
@@ -47,9 +47,17 @@ class MarianMTModel(BartForConditionalGeneration):
|
||||
|
||||
"""
|
||||
config_class = MarianConfig
|
||||
authorized_missing_keys = [
|
||||
"model.encoder.embed_positions.weight",
|
||||
"model.decoder.embed_positions.weight",
|
||||
]
|
||||
keys_to_never_save = [
|
||||
"model.encoder.embed_positions.weight",
|
||||
"model.decoder.embed_positions.weight",
|
||||
]
|
||||
|
||||
def adjust_logits_during_generation(self, logits, cur_len, max_length):
|
||||
logits[:, self.config.pad_token_id] = float("-inf") # never predict pad token.
|
||||
if cur_len == max_length - 1 and self.config.eos_token_id is not None:
|
||||
self._force_token_ids_generation(logits, self.config.eos_token_id)
|
||||
self._force_token_id_to_be_generated(logits, self.config.eos_token_id)
|
||||
return logits
|
||||
|
||||
@@ -29,3 +29,11 @@ class MBartForConditionalGeneration(BartForConditionalGeneration):
|
||||
"""
|
||||
model_type = "mbart"
|
||||
config_class = MBartConfig
|
||||
authorized_missing_keys = [
|
||||
"model.encoder.embed_positions.weight",
|
||||
"model.decoder.embed_positions.weight",
|
||||
]
|
||||
keys_to_never_save = [
|
||||
"model.encoder.embed_positions.weight",
|
||||
"model.decoder.embed_positions.weight",
|
||||
]
|
||||
|
||||
@@ -50,6 +50,10 @@ class PegasusForConditionalGeneration(BartForConditionalGeneration):
|
||||
r"final_logits_bias",
|
||||
r"encoder\.version",
|
||||
r"decoder\.version",
|
||||
r"model.encoder.embed_positions",
|
||||
"model.encoder.embed_positions",
|
||||
"model.decoder.embed_positions",
|
||||
]
|
||||
keys_to_never_save = [
|
||||
"model.encoder.embed_positions.weight",
|
||||
"model.decoder.embed_positions.weight",
|
||||
]
|
||||
|
||||
@@ -1931,14 +1931,21 @@ class ProphetNetForCausalLM(ProphetNetPreTrainedModel):
|
||||
>>> logits = outputs.logits
|
||||
|
||||
>>> # Model can also be used with EncoderDecoder framework
|
||||
>>> from transformers import BertTokenizer, EncoderDecoderModel
|
||||
>>> from transformers import BertTokenizer, EncoderDecoderModel, ProphetNetTokenizer
|
||||
>>> import torch
|
||||
|
||||
>>> tokenizer = BertTokenizer.from_pretrained('bert-uncased-large')
|
||||
>>> model = EncoderDecoderModel.from_encoder_decoder_pretrained("bert-uncased-large", "patrickvonplaten/prophetnet-decoder-clm-large-uncased")
|
||||
>>> tokenizer_enc = BertTokenizer.from_pretrained('bert-large-uncased')
|
||||
>>> tokenizer_dec = ProphetNetTokenizer.from_pretrained('microsoft/prophetnet-large-uncased')
|
||||
>>> model = EncoderDecoderModel.from_encoder_decoder_pretrained("bert-large-uncased", "patrickvonplaten/prophetnet-decoder-clm-large-uncased")
|
||||
|
||||
>>> inputs = tokenizer("Hello, my dog is cute", return_tensors="pt")
|
||||
>>> outputs = model(input_ids=inputs["input_ids"], labels=inputs["input_ids"], return_dict=True)
|
||||
>>> ARTICLE = (
|
||||
... "the us state department said wednesday it had received no "
|
||||
... "formal word from bolivia that it was expelling the us ambassador there "
|
||||
... "but said the charges made against him are `` baseless ."
|
||||
... )
|
||||
>>> input_ids = tokenizer_enc(ARTICLE, return_tensors="pt").input_ids
|
||||
>>> labels = tokenizer_dec("us rejects charges against its ambassador in bolivia", return_tensors="pt").input_ids
|
||||
>>> outputs = model(input_ids=input_ids, decoder_input_ids=labels[:, :-1], labels=labels[:, 1:], return_dict=True)
|
||||
|
||||
>>> loss = outputs.loss
|
||||
"""
|
||||
|
||||
@@ -115,6 +115,11 @@ class RobertaEmbeddings(nn.Module):
|
||||
|
||||
if inputs_embeds is None:
|
||||
inputs_embeds = self.word_embeddings(input_ids)
|
||||
|
||||
max_position_embeddings = self.position_embeddings.num_embeddings
|
||||
if position_ids.max() > max_position_embeddings:
|
||||
raise ValueError("Position ids are too large, the max is {}.".format(max_position_embeddings))
|
||||
|
||||
position_embeddings = self.position_embeddings(position_ids)
|
||||
token_type_embeddings = self.token_type_embeddings(token_type_ids)
|
||||
|
||||
|
||||
@@ -21,6 +21,7 @@ from collections import OrderedDict
|
||||
from .configuration_auto import (
|
||||
AlbertConfig,
|
||||
AutoConfig,
|
||||
BartConfig,
|
||||
BertConfig,
|
||||
CamembertConfig,
|
||||
CTRLConfig,
|
||||
@@ -51,6 +52,7 @@ from .modeling_tf_albert import (
|
||||
TFAlbertForTokenClassification,
|
||||
TFAlbertModel,
|
||||
)
|
||||
from .modeling_tf_bart import TFBartForConditionalGeneration, TFBartModel
|
||||
from .modeling_tf_bert import (
|
||||
TFBertForMaskedLM,
|
||||
TFBertForMultipleChoice,
|
||||
@@ -161,6 +163,7 @@ TF_MODEL_MAPPING = OrderedDict(
|
||||
(T5Config, TFT5Model),
|
||||
(DistilBertConfig, TFDistilBertModel),
|
||||
(AlbertConfig, TFAlbertModel),
|
||||
(BartConfig, TFBartModel),
|
||||
(CamembertConfig, TFCamembertModel),
|
||||
(XLMRobertaConfig, TFXLMRobertaModel),
|
||||
(LongformerConfig, TFLongformerModel),
|
||||
@@ -184,6 +187,7 @@ TF_MODEL_FOR_PRETRAINING_MAPPING = OrderedDict(
|
||||
(T5Config, TFT5ForConditionalGeneration),
|
||||
(DistilBertConfig, TFDistilBertForMaskedLM),
|
||||
(AlbertConfig, TFAlbertForPreTraining),
|
||||
(BartConfig, TFBartForConditionalGeneration),
|
||||
(CamembertConfig, TFCamembertForMaskedLM),
|
||||
(XLMRobertaConfig, TFXLMRobertaForMaskedLM),
|
||||
(RobertaConfig, TFRobertaForMaskedLM),
|
||||
@@ -206,6 +210,7 @@ TF_MODEL_WITH_LM_HEAD_MAPPING = OrderedDict(
|
||||
(T5Config, TFT5ForConditionalGeneration),
|
||||
(DistilBertConfig, TFDistilBertForMaskedLM),
|
||||
(AlbertConfig, TFAlbertForMaskedLM),
|
||||
(BartConfig, TFBartForConditionalGeneration),
|
||||
(CamembertConfig, TFCamembertForMaskedLM),
|
||||
(XLMRobertaConfig, TFXLMRobertaForMaskedLM),
|
||||
(LongformerConfig, TFLongformerForMaskedLM),
|
||||
@@ -256,7 +261,9 @@ TF_MODEL_FOR_MASKED_LM_MAPPING = OrderedDict(
|
||||
]
|
||||
)
|
||||
|
||||
TF_MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING = OrderedDict([(T5Config, TFT5ForConditionalGeneration)])
|
||||
TF_MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING = OrderedDict(
|
||||
[(T5Config, TFT5ForConditionalGeneration), (BartConfig, TFBartForConditionalGeneration)]
|
||||
)
|
||||
|
||||
TF_MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING = OrderedDict(
|
||||
[
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -717,7 +717,7 @@ BERT_START_DOCSTRING = r"""
|
||||
Args:
|
||||
config (:class:`~transformers.BertConfig`): Model configuration class with all the parameters of the model.
|
||||
Initializing with a config file does not load the weights associated with the model, only the configuration.
|
||||
Check out the :meth:`~transformers.PreTrainedModel.from_pretrained` method to load the model weights.
|
||||
Check out the :meth:`~transformers.TFPreTrainedModel.from_pretrained` method to load the model weights.
|
||||
"""
|
||||
|
||||
BERT_INPUTS_DOCSTRING = r"""
|
||||
|
||||
@@ -229,7 +229,7 @@ def load_pytorch_weights_in_tf2_model(tf_model, pt_state_dict, tf_inputs=None, a
|
||||
else:
|
||||
logger.warning(
|
||||
f"All the weights of {tf_model.__class__.__name__} were initialized from the PyTorch model.\n"
|
||||
f"If your task is similar to the task the model of the ckeckpoint was trained on, "
|
||||
f"If your task is similar to the task the model of the checkpoint was trained on, "
|
||||
f"you can already use {tf_model.__class__.__name__} for predictions without further training."
|
||||
)
|
||||
|
||||
@@ -383,7 +383,7 @@ def load_tf2_weights_in_pytorch_model(pt_model, tf_weights, allow_missing_keys=F
|
||||
else:
|
||||
logger.warning(
|
||||
f"All the weights of {pt_model.__class__.__name__} were initialized from the TF 2.0 model.\n"
|
||||
f"If your task is similar to the task the model of the ckeckpoint was trained on, "
|
||||
f"If your task is similar to the task the model of the checkpoint was trained on, "
|
||||
f"you can already use {pt_model.__class__.__name__} for predictions without further training."
|
||||
)
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user