Compare commits
45
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f5438ab8a2 | ||
|
|
ac2c7e398f | ||
|
|
77d6941e64 | ||
|
|
1aca3d6afa | ||
|
|
dc9f245442 | ||
|
|
9a67185344 | ||
|
|
1c1a2ffbff | ||
|
|
07384baf7a | ||
|
|
34334662df | ||
|
|
2f918defa8 | ||
|
|
4d48973523 | ||
|
|
fb650df859 | ||
|
|
c69d19faa8 | ||
|
|
640e6fe190 | ||
|
|
51adb97cd6 | ||
|
|
1551e2dc6d | ||
|
|
ad895af98d | ||
|
|
0b2f46fa9e | ||
|
|
2a7e8e1608 | ||
|
|
e771749777 | ||
|
|
18ecd36f65 | ||
|
|
d018622d8e | ||
|
|
80bdb9c31a | ||
|
|
3caba8d35f | ||
|
|
abc573f51a | ||
|
|
389aba34bf | ||
|
|
ef2d4cd445 | ||
|
|
6ccea0486f | ||
|
|
59da3f2700 | ||
|
|
14c79c3e31 | ||
|
|
ed1845ef4c | ||
|
|
44c340f45f | ||
|
|
c19d04623e | ||
|
|
251eb70c97 | ||
|
|
e4ef57a9bb | ||
|
|
df3f4d2aef | ||
|
|
a9c8bff724 | ||
|
|
b00eb4fb02 | ||
|
|
74daf1f954 | ||
|
|
d6af344c9e | ||
|
|
fa1ddced9e | ||
|
|
6587cf9f84 | ||
|
|
51d9c569fa | ||
|
|
3552d0e0d8 | ||
|
|
29e4597950 |
+335
-19
@@ -63,6 +63,273 @@ references:
|
||||
|
||||
|
||||
jobs:
|
||||
run_tests_torch_and_tf:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch_and_tf-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,tf-cpu,torch,testing,sentencepiece]
|
||||
- run: pip install tapas torch-scatter -f https://pytorch-geometric.com/whl/torch-1.7.0+cpu.html
|
||||
- save_cache:
|
||||
key: v0.4-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: RUN_PT_TF_CROSS_TESTS=1 python -m pytest -n 8 --dist=loadfile -rA -s --make-reports=tests_torch_and_tf ./tests/ -m is_pt_tf_cross_test --durations=0 | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_torch:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,torch,testing,sentencepiece]
|
||||
- run: pip install tapas torch-scatter -f https://pytorch-geometric.com/whl/torch-1.7.0+cpu.html
|
||||
- save_cache:
|
||||
key: v0.4-torch-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -s --make-reports=tests_torch ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_tf:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-tf-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,tf-cpu,testing,sentencepiece]
|
||||
- save_cache:
|
||||
key: v0.4-tf-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -rA -s --make-reports=tests_tf ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_flax:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-flax-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: sudo pip install .[flax,sklearn,torch,testing,sentencepiece]
|
||||
- save_cache:
|
||||
key: v0.4-flax-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -rA -s --make-reports=tests_flax ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_pipelines_torch:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,torch,testing,sentencepiece]
|
||||
- run: pip install tapas torch-scatter -f https://pytorch-geometric.com/whl/torch-1.7.0+cpu.html
|
||||
- save_cache:
|
||||
key: v0.4-torch-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: RUN_PIPELINE_TESTS=1 python -m pytest -n 8 --dist=loadfile -rA -s --make-reports=tests_pipelines_torch -m is_pipeline_test ./tests/ | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_pipelines_tf:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-tf-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,tf-cpu,testing,sentencepiece]
|
||||
- save_cache:
|
||||
key: v0.4-tf-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: RUN_PIPELINE_TESTS=1 python -m pytest -n 8 --dist=loadfile -rA -s --make-reports=tests_pipelines_tf ./tests/ -m is_pipeline_test | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_custom_tokenizers:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
RUN_CUSTOM_TOKENIZERS: yes
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-custom_tokenizers-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[ja,testing,sentencepiece]
|
||||
- run: python -m unidic download
|
||||
- save_cache:
|
||||
key: v0.4-custom_tokenizers-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -s --make-reports=tests_custom_tokenizers ./tests/test_tokenization_bert_japanese.py | tee tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/tests_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_examples_torch:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-torch_examples-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[sklearn,torch,sentencepiece,testing]
|
||||
- run: pip install -r examples/_tests_requirements.txt
|
||||
- save_cache:
|
||||
key: v0.4-torch_examples-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: python -m pytest -n 8 --dist=loadfile -s --make-reports=examples_torch ./examples/ | tee examples_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/examples_output.txt
|
||||
- store_artifacts:
|
||||
path: ~/transformers/reports
|
||||
|
||||
run_tests_git_lfs:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- run: sudo apt-get install git-lfs
|
||||
- run: |
|
||||
git config --global user.email "ci@dummy.com"
|
||||
git config --global user.name "ci"
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install .[testing]
|
||||
- run: RUN_GIT_LFS_TESTS=1 python -m pytest -sv ./tests/test_hf_api.py -k "HfLargefilesTest"
|
||||
|
||||
build_doc:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
steps:
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-build_doc-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run: pip install ."[all, docs]"
|
||||
- save_cache:
|
||||
key: v0.4-build_doc-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: cd docs && make html SPHINXOPTS="-W"
|
||||
- store_artifacts:
|
||||
path: ./docs/_build
|
||||
|
||||
deploy_doc:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
steps:
|
||||
- add_ssh_keys:
|
||||
fingerprints:
|
||||
- "5b:7a:95:18:07:8c:aa:76:4c:60:35:88:ad:60:56:71"
|
||||
- checkout
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v0.4-deploy_doc-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install ."[all,docs]"
|
||||
- save_cache:
|
||||
key: v0.4-deploy_doc-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: ./.circleci/deploy.sh
|
||||
|
||||
check_code_quality:
|
||||
working_directory: ~/transformers
|
||||
docker:
|
||||
@@ -76,22 +343,20 @@ jobs:
|
||||
- v0.4-code_quality-{{ checksum "setup.py" }}
|
||||
- v0.4-{{ checksum "setup.py" }}
|
||||
- run: pip install --upgrade pip
|
||||
- run:
|
||||
command: |
|
||||
set +e
|
||||
echo "my experiment is here"
|
||||
# emulate failure
|
||||
false
|
||||
echo "it can safely fail"
|
||||
some non existing command
|
||||
echo "should still reach here"
|
||||
some non existing command again
|
||||
echo "should still reach here too"
|
||||
false
|
||||
- run:
|
||||
when: always
|
||||
command: |
|
||||
echo "forcing success for this experiment"
|
||||
- run: pip install isort
|
||||
- run: pip install .[all,quality]
|
||||
- save_cache:
|
||||
key: v0.4-code_quality-{{ checksum "setup.py" }}
|
||||
paths:
|
||||
- '~/.cache/pip'
|
||||
- run: black --check examples tests src utils
|
||||
- run: isort --check-only examples tests src utils
|
||||
- run: flake8 examples tests src utils
|
||||
- run: python utils/style_doc.py src/transformers docs/source --max_len 119 --check_only
|
||||
- run: python utils/check_copies.py
|
||||
- run: python utils/check_table.py
|
||||
- run: python utils/check_dummies.py
|
||||
- run: python utils/check_repo.py
|
||||
|
||||
check_repository_consistency:
|
||||
working_directory: ~/transformers
|
||||
@@ -103,7 +368,37 @@ jobs:
|
||||
- checkout
|
||||
- run: pip install requests
|
||||
- run: python ./utils/link_tester.py
|
||||
|
||||
|
||||
# TPU JOBS
|
||||
run_examples_tpu:
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
OMP_NUM_THREADS: 1
|
||||
resource_class: xlarge
|
||||
parallelism: 1
|
||||
steps:
|
||||
- checkout
|
||||
- go/install
|
||||
- *checkout_ml_testing
|
||||
- gcp-gke/install
|
||||
- gcp-gke/update-kubeconfig-with-credentials:
|
||||
cluster: $GKE_CLUSTER
|
||||
perform-login: true
|
||||
- setup_remote_docker
|
||||
- *build_push_docker
|
||||
- *deploy_cluster
|
||||
|
||||
cleanup-gke-jobs:
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
steps:
|
||||
- gcp-gke/install
|
||||
- gcp-gke/update-kubeconfig-with-credentials:
|
||||
cluster: $GKE_CLUSTER
|
||||
perform-login: true
|
||||
- *delete_gke_jobs
|
||||
|
||||
workflow_filters: &workflow_filters
|
||||
filters:
|
||||
branches:
|
||||
@@ -115,5 +410,26 @@ workflows:
|
||||
jobs:
|
||||
- check_code_quality
|
||||
- check_repository_consistency
|
||||
|
||||
|
||||
- run_examples_torch
|
||||
- run_tests_custom_tokenizers
|
||||
- run_tests_torch_and_tf
|
||||
- run_tests_torch
|
||||
- run_tests_tf
|
||||
- run_tests_flax
|
||||
- run_tests_pipelines_torch
|
||||
- run_tests_pipelines_tf
|
||||
- run_tests_git_lfs
|
||||
- build_doc
|
||||
- deploy_doc: *workflow_filters
|
||||
tpu_testing_jobs:
|
||||
triggers:
|
||||
- schedule:
|
||||
# Set to run at the first minute of every hour.
|
||||
cron: "0 8 * * *"
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
- master
|
||||
jobs:
|
||||
- cleanup-gke-jobs
|
||||
- run_examples_tpu
|
||||
|
||||
@@ -50,6 +50,7 @@ jobs:
|
||||
pip install --upgrade pip
|
||||
pip install .[torch,sklearn,testing,onnxruntime,sentencepiece]
|
||||
pip install git+https://github.com/huggingface/datasets
|
||||
pip install pandas torch-scatter -f https://pytorch-geometric.com/whl/torch-1.7.0+cu102.html
|
||||
|
||||
- name: Are GPUs recognized by our DL frameworks
|
||||
run: |
|
||||
@@ -187,6 +188,7 @@ jobs:
|
||||
pip install --upgrade pip
|
||||
pip install .[torch,sklearn,testing,onnxruntime,sentencepiece]
|
||||
pip install git+https://github.com/huggingface/datasets
|
||||
pip install pandas torch-scatter -f https://pytorch-geometric.com/whl/torch-1.7.0+cu102.html
|
||||
|
||||
- name: Are GPUs recognized by our DL frameworks
|
||||
run: |
|
||||
|
||||
@@ -222,6 +222,7 @@ Min, Patrick Lewis, Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih.
|
||||
ultilingual BERT into [DistilmBERT](https://github.com/huggingface/transformers/tree/master/examples/distillation) and a German version of DistilBERT.
|
||||
1. **[SqueezeBert](https://huggingface.co/transformers/model_doc/squeezebert.html)** released with the paper [SqueezeBERT: What can computer vision teach NLP about efficient neural networks?](https://arxiv.org/abs/2006.11316) by Forrest N. Iandola, Albert E. Shaw, Ravi Krishna, and Kurt W. Keutzer.
|
||||
1. **[T5](https://huggingface.co/transformers/model_doc/t5.html)** (from Google AI) released with the paper [Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer](https://arxiv.org/abs/1910.10683) by Colin Raffel and Noam Shazeer and Adam Roberts and Katherine Lee and Sharan Narang and Michael Matena and Yanqi Zhou and Wei Li and Peter J. Liu.
|
||||
1. **[TAPAS](https://huggingface.co/transformers/master/model_doc/tapas.html)** released with the paper [TAPAS: Weakly Supervised Table Parsing via Pre-training](https://arxiv.org/abs/2004.02349) by Jonathan Herzig, Paweł Krzysztof Nowak, Thomas Müller, Francesco Piccinno and Julian Martin Eisenschlos.
|
||||
1. **[Transformer-XL](https://huggingface.co/transformers/model_doc/transformerxl.html)** (from Google/CMU) released with the paper [Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context](https://arxiv.org/abs/1901.02860) by Zihang Dai*, Zhilin Yang*, Yiming Yang, Jaime Carbonell, Quoc V. Le, Ruslan Salakhutdinov.
|
||||
1. **[XLM](https://huggingface.co/transformers/model_doc/xlm.html)** (from Facebook) released together with the paper [Cross-lingual Language Model Pretraining](https://arxiv.org/abs/1901.07291) by Guillaume Lample and Alexis Conneau.
|
||||
1. **[XLM-ProphetNet](https://huggingface.co/transformers/model_doc/xlmprophetnet.html)** (from Microsoft Research) released with the paper [ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training](https://arxiv.org/abs/2001.04063) by Yu Yan, Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang and Ming Zhou.
|
||||
|
||||
@@ -34,5 +34,5 @@ help people access the inner representations, mainly adapted from the great work
|
||||
in https://arxiv.org/abs/1905.10650.
|
||||
|
||||
To help you understand and use these features, we have added a specific example script: `bertology.py
|
||||
<https://github.com/huggingface/transformers/blob/master/examples/bertology/run_bertology.py>`_ while extract
|
||||
information and prune a model pre-trained on GLUE.
|
||||
<https://github.com/huggingface/transformers/blob/master/examples/research_projects/bertology/run_bertology.py>`_ while
|
||||
extract information and prune a model pre-trained on GLUE.
|
||||
|
||||
+1
-1
@@ -26,7 +26,7 @@ author = u'huggingface'
|
||||
# The short X.Y version
|
||||
version = u''
|
||||
# The full version, including alpha/beta/rc tags
|
||||
release = u'4.0.0'
|
||||
release = u'4.1.0'
|
||||
|
||||
|
||||
# -- General configuration ---------------------------------------------------
|
||||
|
||||
+11
-5
@@ -176,19 +176,22 @@ and conversion utilities for the following models:
|
||||
30. :doc:`T5 <model_doc/t5>` (from Google AI) released with the paper `Exploring the Limits of Transfer Learning with a
|
||||
Unified Text-to-Text Transformer <https://arxiv.org/abs/1910.10683>`__ by Colin Raffel and Noam Shazeer and Adam
|
||||
Roberts and Katherine Lee and Sharan Narang and Michael Matena and Yanqi Zhou and Wei Li and Peter J. Liu.
|
||||
31. :doc:`Transformer-XL <model_doc/transformerxl>` (from Google/CMU) released with the paper `Transformer-XL:
|
||||
31. `TAPAS <https://huggingface.co/transformers/master/model_doc/tapas.html>`__ released with the paper `TAPAS: Weakly
|
||||
Supervised Table Parsing via Pre-training <https://arxiv.org/abs/2004.02349>`__ by Jonathan Herzig, Paweł Krzysztof
|
||||
Nowak, Thomas Müller, Francesco Piccinno and Julian Martin Eisenschlos.
|
||||
32. :doc:`Transformer-XL <model_doc/transformerxl>` (from Google/CMU) released with the paper `Transformer-XL:
|
||||
Attentive Language Models Beyond a Fixed-Length Context <https://arxiv.org/abs/1901.02860>`__ by Zihang Dai*,
|
||||
Zhilin Yang*, Yiming Yang, Jaime Carbonell, Quoc V. Le, Ruslan Salakhutdinov.
|
||||
32. :doc:`XLM <model_doc/xlm>` (from Facebook) released together with the paper `Cross-lingual Language Model
|
||||
33. :doc:`XLM <model_doc/xlm>` (from Facebook) released together with the paper `Cross-lingual Language Model
|
||||
Pretraining <https://arxiv.org/abs/1901.07291>`__ by Guillaume Lample and Alexis Conneau.
|
||||
33. :doc:`XLM-ProphetNet <model_doc/xlmprophetnet>` (from Microsoft Research) released with the paper `ProphetNet:
|
||||
34. :doc:`XLM-ProphetNet <model_doc/xlmprophetnet>` (from Microsoft Research) released with the paper `ProphetNet:
|
||||
Predicting Future N-gram for Sequence-to-Sequence Pre-training <https://arxiv.org/abs/2001.04063>`__ by Yu Yan,
|
||||
Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang and Ming Zhou.
|
||||
34. :doc:`XLM-RoBERTa <model_doc/xlmroberta>` (from Facebook AI), released together with the paper `Unsupervised
|
||||
35. :doc:`XLM-RoBERTa <model_doc/xlmroberta>` (from Facebook AI), released together with the paper `Unsupervised
|
||||
Cross-lingual Representation Learning at Scale <https://arxiv.org/abs/1911.02116>`__ by Alexis Conneau*, Kartikay
|
||||
Khandelwal*, Naman Goyal, Vishrav Chaudhary, Guillaume Wenzek, Francisco Guzmán, Edouard Grave, Myle Ott, Luke
|
||||
Zettlemoyer and Veselin Stoyanov.
|
||||
35. :doc:`XLNet <model_doc/xlnet>` (from Google/CMU) released with the paper `XLNet: Generalized Autoregressive
|
||||
36. :doc:`XLNet <model_doc/xlnet>` (from Google/CMU) released with the paper `XLNet: Generalized Autoregressive
|
||||
Pretraining for Language Understanding <https://arxiv.org/abs/1906.08237>`__ by Zhilin Yang*, Zihang Dai*, Yiming
|
||||
Yang, Jaime Carbonell, Ruslan Salakhutdinov, Quoc V. Le.
|
||||
|
||||
@@ -269,6 +272,8 @@ TensorFlow and/or Flax.
|
||||
+-----------------------------+----------------+----------------+-----------------+--------------------+--------------+
|
||||
| T5 | ✅ | ✅ | ✅ | ✅ | ❌ |
|
||||
+-----------------------------+----------------+----------------+-----------------+--------------------+--------------+
|
||||
| TAPAS | ✅ | ❌ | ✅ | ❌ | ❌ |
|
||||
+-----------------------------+----------------+----------------+-----------------+--------------------+--------------+
|
||||
| Transformer-XL | ✅ | ❌ | ✅ | ✅ | ❌ |
|
||||
+-----------------------------+----------------+----------------+-----------------+--------------------+--------------+
|
||||
| XLM | ✅ | ❌ | ✅ | ✅ | ❌ |
|
||||
@@ -382,6 +387,7 @@ TensorFlow and/or Flax.
|
||||
model_doc/roberta
|
||||
model_doc/squeezebert
|
||||
model_doc/t5
|
||||
model_doc/tapas
|
||||
model_doc/transformerxl
|
||||
model_doc/xlm
|
||||
model_doc/xlmprophetnet
|
||||
|
||||
@@ -91,8 +91,6 @@ TensorFlow loss functions
|
||||
TensorFlow Helper Functions
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autofunction:: transformers.modeling_tf_utils.cast_bool_to_primitive
|
||||
|
||||
.. autofunction:: transformers.modeling_tf_utils.get_initializer
|
||||
|
||||
.. autofunction:: transformers.modeling_tf_utils.keras_serializable
|
||||
|
||||
@@ -13,9 +13,10 @@
|
||||
Models
|
||||
-----------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
The base classes :class:`~transformers.PreTrainedModel` and :class:`~transformers.TFPreTrainedModel` implement the
|
||||
common methods for loading/saving a model either from a local file or directory, or from a pretrained model
|
||||
configuration provided by the library (downloaded from HuggingFace's AWS S3 repository).
|
||||
The base classes :class:`~transformers.PreTrainedModel`, :class:`~transformers.TFPreTrainedModel`, and
|
||||
:class:`~transformers.FlaxPreTrainedModel` implement the common methods for loading/saving a model either from a local
|
||||
file or directory, or from a pretrained model configuration provided by the library (downloaded from HuggingFace's AWS
|
||||
S3 repository).
|
||||
|
||||
:class:`~transformers.PreTrainedModel` and :class:`~transformers.TFPreTrainedModel` also implement a few methods which
|
||||
are common among all the models to:
|
||||
@@ -57,6 +58,13 @@ TFModelUtilsMixin
|
||||
:members:
|
||||
|
||||
|
||||
FlaxPreTrainedModel
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.FlaxPreTrainedModel
|
||||
:members:
|
||||
|
||||
|
||||
Generation
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
|
||||
@@ -34,6 +34,7 @@ There are two categories of pipeline abstractions to be aware about:
|
||||
- :class:`~transformers.TranslationPipeline`
|
||||
- :class:`~transformers.ZeroShotClassificationPipeline`
|
||||
- :class:`~transformers.Text2TextGenerationPipeline`
|
||||
- :class:`~transformers.TableQuestionAnsweringPipeline`
|
||||
|
||||
The pipeline abstraction
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
@@ -91,6 +92,13 @@ SummarizationPipeline
|
||||
:special-members: __call__
|
||||
:members:
|
||||
|
||||
TableQuestionAnsweringPipeline
|
||||
=======================================================================================================================
|
||||
|
||||
.. autoclass:: transformers.TableQuestionAnsweringPipeline
|
||||
:special-members: __call__
|
||||
|
||||
|
||||
TextClassificationPipeline
|
||||
=======================================================================================================================
|
||||
|
||||
|
||||
@@ -114,6 +114,13 @@ AutoModelForQuestionAnswering
|
||||
:members:
|
||||
|
||||
|
||||
AutoModelForTableQuestionAnswering
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.AutoModelForTableQuestionAnswering
|
||||
:members:
|
||||
|
||||
|
||||
TFAutoModel
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
|
||||
@@ -138,3 +138,9 @@ TFOpenAIGPTDoubleHeadsModel
|
||||
|
||||
.. autoclass:: transformers.TFOpenAIGPTDoubleHeadsModel
|
||||
:members: call
|
||||
|
||||
TFOpenAIGPTForSequenceClassification
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TFOpenAIGPTForSequenceClassification
|
||||
:members: call
|
||||
|
||||
@@ -34,6 +34,12 @@ contrast to most prior work, we also pretrain Longformer and finetune it on a va
|
||||
pretrained Longformer consistently outperforms RoBERTa on long document tasks and sets new state-of-the-art results on
|
||||
WikiHop and TriviaQA.*
|
||||
|
||||
Tips:
|
||||
|
||||
- Since the Longformer is based on RoBERTa, it doesn't have :obj:`token_type_ids`. You don't need to indicate which
|
||||
token belongs to which segment. Just separate your segments with the separation token :obj:`tokenizer.sep_token` (or
|
||||
:obj:`</s>`).
|
||||
|
||||
The Authors' code can be found `here <https://github.com/allenai/longformer>`__.
|
||||
|
||||
Longformer Self Attention
|
||||
|
||||
@@ -131,7 +131,7 @@ T5EncoderModel
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.T5EncoderModel
|
||||
:members: forward
|
||||
:members: forward, parallelize, deparallelize
|
||||
|
||||
TFT5Model
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
@@ -0,0 +1,434 @@
|
||||
TAPAS
|
||||
-----------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
.. note::
|
||||
|
||||
This is a recently introduced model so the API hasn't been tested extensively. There may be some bugs or slight
|
||||
breaking changes to fix them in the future.
|
||||
|
||||
|
||||
|
||||
Overview
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
The TAPAS model was proposed in `TAPAS: Weakly Supervised Table Parsing via Pre-training
|
||||
<https://www.aclweb.org/anthology/2020.acl-main.398>`__ by Jonathan Herzig, Paweł Krzysztof Nowak, Thomas Müller,
|
||||
Francesco Piccinno and Julian Martin Eisenschlos. It's a BERT-based model specifically designed (and pre-trained) for
|
||||
answering questions about tabular data. Compared to BERT, TAPAS uses relative position embeddings and has 7 token types
|
||||
that encode tabular structure. TAPAS is pre-trained on the masked language modeling (MLM) objective on a large dataset
|
||||
comprising millions of tables from English Wikipedia and corresponding texts. For question answering, TAPAS has 2 heads
|
||||
on top: a cell selection head and an aggregation head, for (optionally) performing aggregations (such as counting or
|
||||
summing) among selected cells. TAPAS has been fine-tuned on several datasets: `SQA
|
||||
<https://www.microsoft.com/en-us/download/details.aspx?id=54253>`__ (Sequential Question Answering by Microsoft), `WTQ
|
||||
<https://github.com/ppasupat/WikiTableQuestions>`__ (Wiki Table Questions by Stanford University) and `WikiSQL
|
||||
<https://github.com/salesforce/WikiSQL>`__ (by Salesforce). It achieves state-of-the-art on both SQA and WTQ, while
|
||||
having comparable performance to SOTA on WikiSQL, with a much simpler architecture.
|
||||
|
||||
The abstract from the paper is the following:
|
||||
|
||||
*Answering natural language questions over tables is usually seen as a semantic parsing task. To alleviate the
|
||||
collection cost of full logical forms, one popular approach focuses on weak supervision consisting of denotations
|
||||
instead of logical forms. However, training semantic parsers from weak supervision poses difficulties, and in addition,
|
||||
the generated logical forms are only used as an intermediate step prior to retrieving the denotation. In this paper, we
|
||||
present TAPAS, an approach to question answering over tables without generating logical forms. TAPAS trains from weak
|
||||
supervision, and predicts the denotation by selecting table cells and optionally applying a corresponding aggregation
|
||||
operator to such selection. TAPAS extends BERT's architecture to encode tables as input, initializes from an effective
|
||||
joint pre-training of text segments and tables crawled from Wikipedia, and is trained end-to-end. We experiment with
|
||||
three different semantic parsing datasets, and find that TAPAS outperforms or rivals semantic parsing models by
|
||||
improving state-of-the-art accuracy on SQA from 55.1 to 67.2 and performing on par with the state-of-the-art on WIKISQL
|
||||
and WIKITQ, but with a simpler model architecture. We additionally find that transfer learning, which is trivial in our
|
||||
setting, from WIKISQL to WIKITQ, yields 48.7 accuracy, 4.2 points above the state-of-the-art.*
|
||||
|
||||
In addition, the authors have further pre-trained TAPAS to recognize **table entailment**, by creating a balanced
|
||||
dataset of millions of automatically created training examples which are learned in an intermediate step prior to
|
||||
fine-tuning. The authors of TAPAS call this further pre-training intermediate pre-training (since TAPAS is first
|
||||
pre-trained on MLM, and then on another dataset). They found that intermediate pre-training further improves
|
||||
performance on SQA, achieving a new state-of-the-art as well as state-of-the-art on `TabFact
|
||||
<https://github.com/wenhuchen/Table-Fact-Checking>`__, a large-scale dataset with 16k Wikipedia tables for table
|
||||
entailment (a binary classification task). For more details, see their follow-up paper: `Understanding tables with
|
||||
intermediate pre-training <https://www.aclweb.org/anthology/2020.findings-emnlp.27/>`__ by Julian Martin Eisenschlos,
|
||||
Syrine Krichene and Thomas Müller.
|
||||
|
||||
The original code can be found `here <https://github.com/google-research/tapas>`__.
|
||||
|
||||
Tips:
|
||||
|
||||
- TAPAS is a model that uses relative position embeddings by default (restarting the position embeddings at every cell
|
||||
of the table). Note that this is something that was added after the publication of the original TAPAS paper.
|
||||
According to the authors, this usually results in a slightly better performance, and allows you to encode longer
|
||||
sequences without running out of embeddings. This is reflected in the ``reset_position_index_per_cell`` parameter of
|
||||
:class:`~transformers.TapasConfig`, which is set to ``True`` by default. The default versions of the models available
|
||||
in the `model hub <https://huggingface.co/models?search=tapas>`_ all use relative position embeddings. You can still
|
||||
use the ones with absolute position embeddings by passing in an additional argument ``revision="no_reset"`` when
|
||||
calling the ``.from_pretrained()`` method. Note that it's usually advised to pad the inputs on the right rather than
|
||||
the left.
|
||||
- TAPAS is based on BERT, so ``TAPAS-base`` for example corresponds to a ``BERT-base`` architecture. Of course,
|
||||
TAPAS-large will result in the best performance (the results reported in the paper are from TAPAS-large). Results of
|
||||
the various sized models are shown on the `original Github repository <https://github.com/google-research/tapas>`_.
|
||||
- TAPAS has checkpoints fine-tuned on SQA, which are capable of answering questions related to a table in a
|
||||
conversational set-up. This means that you can ask follow-up questions such as "what is his age?" related to the
|
||||
previous question. Note that the forward pass of TAPAS is a bit different in case of a conversational set-up: in that
|
||||
case, you have to feed every table-question pair one by one to the model, such that the `prev_labels` token type ids
|
||||
can be overwritten by the predicted `labels` of the model to the previous question. See "Usage" section for more
|
||||
info.
|
||||
- TAPAS is similar to BERT and therefore relies on the masked language modeling (MLM) objective. It is therefore
|
||||
efficient at predicting masked tokens and at NLU in general, but is not optimal for text generation. Models trained
|
||||
with a causal language modeling (CLM) objective are better in that regard.
|
||||
|
||||
|
||||
Usage: fine-tuning
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
Here we explain how you can fine-tune :class:`~transformers.TapasForQuestionAnswering` on your own dataset.
|
||||
|
||||
**STEP 1: Choose one of the 3 ways in which you can use TAPAS - or experiment**
|
||||
|
||||
Basically, there are 3 different ways in which one can fine-tune :class:`~transformers.TapasForQuestionAnswering`,
|
||||
corresponding to the different datasets on which Tapas was fine-tuned:
|
||||
|
||||
1. SQA: if you're interested in asking follow-up questions related to a table, in a conversational set-up. For example
|
||||
if you first ask "what's the name of the first actor?" then you can ask a follow-up question such as "how old is
|
||||
he?". Here, questions do not involve any aggregation (all questions are cell selection questions).
|
||||
2. WTQ: if you're not interested in asking questions in a conversational set-up, but rather just asking questions
|
||||
related to a table, which might involve aggregation, such as counting a number of rows, summing up cell values or
|
||||
averaging cell values. You can then for example ask "what's the total number of goals Cristiano Ronaldo made in his
|
||||
career?". This case is also called **weak supervision**, since the model itself must learn the appropriate
|
||||
aggregation operator (SUM/COUNT/AVERAGE/NONE) given only the answer to the question as supervision.
|
||||
3. WikiSQL-supervised: this dataset is based on WikiSQL with the model being given the ground truth aggregation
|
||||
operator during training. This is also called **strong supervision**. Here, learning the appropriate aggregation
|
||||
operator is much easier.
|
||||
|
||||
To summarize:
|
||||
|
||||
+------------------------------------+----------------------+-------------------------------------------------------------------------------------------------------------------+
|
||||
| **Task** | **Example dataset** | **Description** |
|
||||
+------------------------------------+----------------------+-------------------------------------------------------------------------------------------------------------------+
|
||||
| Conversational | SQA | Conversational, only cell selection questions |
|
||||
+------------------------------------+----------------------+-------------------------------------------------------------------------------------------------------------------+
|
||||
| Weak supervision for aggregation | WTQ | Questions might involve aggregation, and the model must learn this given only the answer as supervision |
|
||||
+------------------------------------+----------------------+-------------------------------------------------------------------------------------------------------------------+
|
||||
| Strong supervision for aggregation | WikiSQL-supervised | Questions might involve aggregation, and the model must learn this given the gold aggregation operator |
|
||||
+------------------------------------+----------------------+-------------------------------------------------------------------------------------------------------------------+
|
||||
|
||||
Initializing a model with a pre-trained base and randomly initialized classification heads from the model hub can be
|
||||
done as follows (be sure to have installed the `torch-scatter dependency <https://github.com/rusty1s/pytorch_scatter>`_
|
||||
for your environment):
|
||||
|
||||
.. code-block::
|
||||
|
||||
>>> from transformers import TapasConfig, TapasForQuestionAnswering
|
||||
|
||||
>>> # for example, the base sized model with default SQA configuration
|
||||
>>> model = TapasForQuestionAnswering.from_pretrained('google/tapas-base')
|
||||
|
||||
>>> # or, the base sized model with WTQ configuration
|
||||
>>> config = TapasConfig.from_pretrained('google/tapas-base-finetuned-wtq')
|
||||
>>> model = TapasForQuestionAnswering.from_pretrained('google/tapas-base', config=config)
|
||||
|
||||
>>> # or, the base sized model with WikiSQL configuration
|
||||
>>> config = TapasConfig('google-base-finetuned-wikisql-supervised')
|
||||
>>> model = TapasForQuestionAnswering.from_pretrained('google/tapas-base', config=config)
|
||||
|
||||
|
||||
Of course, you don't necessarily have to follow one of these three ways in which TAPAS was fine-tuned. You can also
|
||||
experiment by defining any hyperparameters you want when initializing :class:`~transformers.TapasConfig`, and then
|
||||
create a :class:`~transformers.TapasForQuestionAnswering` based on that configuration. For example, if you have a
|
||||
dataset that has both conversational questions and questions that might involve aggregation, then you can do it this
|
||||
way. Here's an example:
|
||||
|
||||
.. code-block::
|
||||
|
||||
>>> from transformers import TapasConfig, TapasForQuestionAnswering
|
||||
|
||||
>>> # you can initialize the classification heads any way you want (see docs of TapasConfig)
|
||||
>>> config = TapasConfig(num_aggregation_labels=3, average_logits_per_cell=True, select_one_column=False)
|
||||
>>> # initializing the pre-trained base sized model with our custom classification heads
|
||||
>>> model = TapasForQuestionAnswering.from_pretrained('google/tapas-base', config=config)
|
||||
|
||||
What you can also do is start from an already fine-tuned checkpoint. A note here is that the already fine-tuned
|
||||
checkpoint on WTQ has some issues due to the L2-loss which is somewhat brittle. See `here
|
||||
<https://github.com/google-research/tapas/issues/91#issuecomment-735719340>`__ for more info.
|
||||
|
||||
For a list of all pre-trained and fine-tuned TAPAS checkpoints available in the HuggingFace model hub, see `here
|
||||
<https://huggingface.co/models?search=tapas>`__.
|
||||
|
||||
**STEP 2: Prepare your data in the SQA format**
|
||||
|
||||
Second, no matter what you picked above, you should prepare your dataset in the `SQA format
|
||||
<https://www.microsoft.com/en-us/download/details.aspx?id=54253>`__. This format is a TSV/CSV file with the following
|
||||
columns:
|
||||
|
||||
- ``id``: optional, id of the table-question pair, for bookkeeping purposes.
|
||||
- ``annotator``: optional, id of the person who annotated the table-question pair, for bookkeeping purposes.
|
||||
- ``position``: integer indicating if the question is the first, second, third,... related to the table. Only required
|
||||
in case of conversational setup (SQA). You don't need this column in case you're going for WTQ/WikiSQL-supervised.
|
||||
- ``question``: string
|
||||
- ``table_file``: string, name of a csv file containing the tabular data
|
||||
- ``answer_coordinates``: list of one or more tuples (each tuple being a cell coordinate, i.e. row, column pair that is
|
||||
part of the answer)
|
||||
- ``answer_text``: list of one or more strings (each string being a cell value that is part of the answer)
|
||||
- ``aggregation_label``: index of the aggregation operator. Only required in case of strong supervision for aggregation
|
||||
(the WikiSQL-supervised case)
|
||||
- ``float_answer``: the float answer to the question, if there is one (np.nan if there isn't). Only required in case of
|
||||
weak supervision for aggregation (such as WTQ and WikiSQL)
|
||||
|
||||
The tables themselves should be present in a folder, each table being a separate csv file. Note that the authors of the
|
||||
TAPAS algorithm used conversion scripts with some automated logic to convert the other datasets (WTQ, WikiSQL) into the
|
||||
SQA format. The author explains this `here
|
||||
<https://github.com/google-research/tapas/issues/50#issuecomment-705465960>`__. Interestingly, these conversion scripts
|
||||
are not perfect (the ``answer_coordinates`` and ``float_answer`` fields are populated based on the ``answer_text``),
|
||||
meaning that WTQ and WikiSQL results could actually be improved.
|
||||
|
||||
**STEP 3: Convert your data into PyTorch tensors using TapasTokenizer**
|
||||
|
||||
Third, given that you've prepared your data in this TSV/CSV format (and corresponding CSV files containing the tabular
|
||||
data), you can then use :class:`~transformers.TapasTokenizer` to convert table-question pairs into :obj:`input_ids`,
|
||||
:obj:`attention_mask`, :obj:`token_type_ids` and so on. Again, based on which of the three cases you picked above,
|
||||
:class:`~transformers.TapasForQuestionAnswering` requires different inputs to be fine-tuned:
|
||||
|
||||
+------------------------------------+----------------------------------------------------------------------------------------------+
|
||||
| **Task** | **Required inputs** |
|
||||
+------------------------------------+----------------------------------------------------------------------------------------------+
|
||||
| Conversational | ``input_ids``, ``attention_mask``, ``token_type_ids``, ``labels`` |
|
||||
+------------------------------------+----------------------------------------------------------------------------------------------+
|
||||
| Weak supervision for aggregation | ``input_ids``, ``attention_mask``, ``token_type_ids``, ``labels``, ``numeric_values``, |
|
||||
| | ``numeric_values_scale``, ``float_answer`` |
|
||||
+------------------------------------+----------------------------------------------------------------------------------------------+
|
||||
| Strong supervision for aggregation | ``input ids``, ``attention mask``, ``token type ids``, ``labels``, ``aggregation_labels`` |
|
||||
+------------------------------------+----------------------------------------------------------------------------------------------+
|
||||
|
||||
:class:`~transformers.TapasTokenizer` creates the ``labels``, ``numeric_values`` and ``numeric_values_scale`` based on
|
||||
the ``answer_coordinates`` and ``answer_text`` columns of the TSV file. The ``float_answer`` and ``aggregation_labels``
|
||||
are already in the TSV file of step 2. Here's an example:
|
||||
|
||||
.. code-block::
|
||||
|
||||
>>> from transformers import TapasTokenizer
|
||||
>>> import pandas as pd
|
||||
|
||||
>>> model_name = 'google/tapas-base'
|
||||
>>> tokenizer = TapasTokenizer.from_pretrained(model_name)
|
||||
|
||||
>>> data = {'Actors': ["Brad Pitt", "Leonardo Di Caprio", "George Clooney"], 'Number of movies': ["87", "53", "69"]}
|
||||
>>> queries = ["What is the name of the first actor?", "How many movies has George Clooney played in?", "What is the total number of movies?"]
|
||||
>>> answer_coordinates = [[(0, 0)], [(2, 1)], [(0, 1), (1, 1), (2, 1)]]
|
||||
>>> answer_text = [["Brad Pitt"], ["69"], ["209"]]
|
||||
>>> table = pd.DataFrame.from_dict(data)
|
||||
>>> inputs = tokenizer(table=table, queries=queries, answer_coordinates=answer_coordinates, answer_text=answer_text, padding='max_length', return_tensors='pt')
|
||||
>>> inputs
|
||||
{'input_ids': tensor([[ ... ]]), 'attention_mask': tensor([[...]]), 'token_type_ids': tensor([[[...]]]),
|
||||
'numeric_values': tensor([[ ... ]]), 'numeric_values_scale: tensor([[ ... ]]), labels: tensor([[ ... ]])}
|
||||
|
||||
Note that :class:`~transformers.TapasTokenizer` expects the data of the table to be **text-only**. You can use
|
||||
``.astype(str)`` on a dataframe to turn it into text-only data. Of course, this only shows how to encode a single
|
||||
training example. It is advised to create a PyTorch dataset and a corresponding dataloader:
|
||||
|
||||
.. code-block::
|
||||
|
||||
>>> import torch
|
||||
>>> import pandas as pd
|
||||
|
||||
>>> tsv_path = "your_path_to_the_tsv_file"
|
||||
>>> table_csv_path = "your_path_to_a_directory_containing_all_csv_files"
|
||||
|
||||
>>> class TableDataset(torch.utils.data.Dataset):
|
||||
... def __init__(self, data, tokenizer):
|
||||
... self.data = data
|
||||
... self.tokenizer = tokenizer
|
||||
...
|
||||
... def __getitem__(self, idx):
|
||||
... item = data.iloc[idx]
|
||||
... table = pd.read_csv(table_csv_path + item.table_file).astype(str) # be sure to make your table data text only
|
||||
... encoding = self.tokenizer(table=table,
|
||||
... queries=item.question,
|
||||
... answer_coordinates=item.answer_coordinates,
|
||||
... answer_text=item.answer_text,
|
||||
... truncation=True,
|
||||
... padding="max_length",
|
||||
... return_tensors="pt"
|
||||
... )
|
||||
... # remove the batch dimension which the tokenizer adds by default
|
||||
... encoding = {key: val.squeeze(0) for key, val in encoding.items()}
|
||||
... # add the float_answer which is also required (weak supervision for aggregation case)
|
||||
... encoding["float_answer"] = torch.tensor(item.float_answer)
|
||||
... return encoding
|
||||
...
|
||||
... def __len__(self):
|
||||
... return len(self.data)
|
||||
|
||||
>>> data = pd.read_csv(tsv_path, sep='\t')
|
||||
>>> train_dataset = TableDataset(data, tokenizer)
|
||||
>>> train_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=32)
|
||||
|
||||
Note that here, we encode each table-question pair independently. This is fine as long as your dataset is **not
|
||||
conversational**. In case your dataset involves conversational questions (such as in SQA), then you should first group
|
||||
together the ``queries``, ``answer_coordinates`` and ``answer_text`` per table (in the order of their ``position``
|
||||
index) and batch encode each table with its questions. This will make sure that the ``prev_labels`` token types (see
|
||||
docs of :class:`~transformers.TapasTokenizer`) are set correctly. See `this notebook
|
||||
<https://github.com/NielsRogge/Transformers-Tutorials/blob/master/Fine_tuning_TapasForQuestionAnswering_on_SQA.ipynb>`__
|
||||
for more info.
|
||||
|
||||
**STEP 4: Train (fine-tune) TapasForQuestionAnswering**
|
||||
|
||||
You can then fine-tune :class:`~transformers.TapasForQuestionAnswering` using native PyTorch as follows (shown here for
|
||||
the weak supervision for aggregation case):
|
||||
|
||||
.. code-block::
|
||||
|
||||
>>> from transformers import TapasConfig, TapasForQuestionAnswering, AdamW
|
||||
|
||||
>>> # this is the default WTQ configuration
|
||||
>>> config = TapasConfig(
|
||||
... num_aggregation_labels = 4,
|
||||
... use_answer_as_supervision = True,
|
||||
... answer_loss_cutoff = 0.664694,
|
||||
... cell_selection_preference = 0.207951,
|
||||
... huber_loss_delta = 0.121194,
|
||||
... init_cell_selection_weights_to_zero = True,
|
||||
... select_one_column = True,
|
||||
... allow_empty_column_selection = False,
|
||||
... temperature = 0.0352513,
|
||||
... )
|
||||
>>> model = TapasForQuestionAnswering.from_pretrained("google/tapas-base", config=config)
|
||||
|
||||
>>> optimizer = AdamW(model.parameters(), lr=5e-5)
|
||||
|
||||
>>> for epoch in range(2): # loop over the dataset multiple times
|
||||
... for idx, batch in enumerate(train_dataloader):
|
||||
... # get the inputs;
|
||||
... input_ids = batch["input_ids"]
|
||||
... attention_mask = batch["attention_mask"]
|
||||
... token_type_ids = batch["token_type_ids"]
|
||||
... labels = batch["labels"]
|
||||
... numeric_values = batch["numeric_values"]
|
||||
... numeric_values_scale = batch["numeric_values_scale"]
|
||||
... float_answer = batch["float_answer"]
|
||||
|
||||
... # zero the parameter gradients
|
||||
... optimizer.zero_grad()
|
||||
|
||||
... # forward + backward + optimize
|
||||
... outputs = model(input_ids=input_ids, attention_mask=attention_mask, token_type_ids=token_type_ids,
|
||||
... labels=labels, numeric_values=numeric_values, numeric_values_scale=numeric_values_scale,
|
||||
... float_answer=float_answer)
|
||||
... loss = outputs.loss
|
||||
... loss.backward()
|
||||
... optimizer.step()
|
||||
|
||||
Usage: inference
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
Here we explain how you can use :class:`~transformers.TapasForQuestionAnswering` for inference (i.e. making predictions
|
||||
on new data). For inference, only ``input_ids``, ``attention_mask`` and ``token_type_ids`` (which you can obtain using
|
||||
:class:`~transformers.TapasTokenizer`) have to be provided to the model to obtain the logits. Next, you can use the
|
||||
handy ``convert_logits_to_predictions`` method of :class:`~transformers.TapasTokenizer` to convert these into predicted
|
||||
coordinates and optional aggregation indices.
|
||||
|
||||
However, note that inference is **different** depending on whether or not the setup is conversational. In a
|
||||
non-conversational set-up, inference can be done in parallel on all table-question pairs of a batch. Here's an example
|
||||
of that:
|
||||
|
||||
.. code-block::
|
||||
|
||||
>>> from transformers import TapasTokenizer, TapasForQuestionAnswering
|
||||
>>> import pandas as pd
|
||||
|
||||
>>> model_name = 'google/tapas-base-finetuned-wtq'
|
||||
>>> model = TapasForQuestionAnswering.from_pretrained(model_name)
|
||||
>>> tokenizer = TapasTokenizer.from_pretrained(model_name)
|
||||
|
||||
>>> data = {'Actors': ["Brad Pitt", "Leonardo Di Caprio", "George Clooney"], 'Number of movies': ["87", "53", "69"]}
|
||||
>>> queries = ["What is the name of the first actor?", "How many movies has George Clooney played in?", "What is the total number of movies?"]
|
||||
>>> table = pd.DataFrame.from_dict(data)
|
||||
>>> inputs = tokenizer(table=table, queries=queries, padding='max_length', return_tensors="pt")
|
||||
>>> outputs = model(**inputs)
|
||||
>>> predicted_answer_coordinates, predicted_aggregation_indices = tokenizer.convert_logits_to_predictions(
|
||||
... inputs,
|
||||
... outputs.logits.detach(),
|
||||
... outputs.logits_aggregation.detach()
|
||||
...)
|
||||
|
||||
>>> # let's print out the results:
|
||||
>>> id2aggregation = {0: "NONE", 1: "SUM", 2: "AVERAGE", 3:"COUNT"}
|
||||
>>> aggregation_predictions_string = [id2aggregation[x] for x in predicted_aggregation_indices]
|
||||
|
||||
>>> answers = []
|
||||
>>> for coordinates in predicted_answer_coordinates:
|
||||
... if len(coordinates) == 1:
|
||||
... # only a single cell:
|
||||
... answers.append(table.iat[coordinates[0]])
|
||||
... else:
|
||||
... # multiple cells
|
||||
... cell_values = []
|
||||
... for coordinate in coordinates:
|
||||
... cell_values.append(table.iat[coordinate])
|
||||
... answers.append(", ".join(cell_values))
|
||||
|
||||
>>> display(table)
|
||||
>>> print("")
|
||||
>>> for query, answer, predicted_agg in zip(queries, answers, aggregation_predictions_string):
|
||||
... print(query)
|
||||
... if predicted_agg == "NONE":
|
||||
... print("Predicted answer: " + answer)
|
||||
... else:
|
||||
... print("Predicted answer: " + predicted_agg + " > " + answer)
|
||||
What is the name of the first actor?
|
||||
Predicted answer: Brad Pitt
|
||||
How many movies has George Clooney played in?
|
||||
Predicted answer: COUNT > 69
|
||||
What is the total number of movies?
|
||||
Predicted answer: SUM > 87, 53, 69
|
||||
|
||||
In case of a conversational set-up, then each table-question pair must be provided **sequentially** to the model, such
|
||||
that the ``prev_labels`` token types can be overwritten by the predicted ``labels`` of the previous table-question
|
||||
pair. Again, more info can be found in `this notebook
|
||||
<https://github.com/NielsRogge/Transformers-Tutorials/blob/master/Fine_tuning_TapasForQuestionAnswering_on_SQA.ipynb>`__.
|
||||
|
||||
|
||||
Tapas specific outputs
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.models.tapas.modeling_tapas.TableQuestionAnsweringOutput
|
||||
:members:
|
||||
|
||||
|
||||
TapasConfig
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TapasConfig
|
||||
:members:
|
||||
|
||||
|
||||
TapasTokenizer
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TapasTokenizer
|
||||
:members: __call__, convert_logits_to_predictions, save_vocabulary
|
||||
|
||||
|
||||
TapasModel
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TapasModel
|
||||
:members: forward
|
||||
|
||||
|
||||
TapasForMaskedLM
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TapasForMaskedLM
|
||||
:members: forward
|
||||
|
||||
|
||||
TapasForSequenceClassification
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TapasForSequenceClassification
|
||||
:members: forward
|
||||
|
||||
|
||||
TapasForQuestionAnswering
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
.. autoclass:: transformers.TapasForQuestionAnswering
|
||||
:members: forward
|
||||
@@ -60,7 +60,7 @@ Basic steps
|
||||
In order to upload a model, you'll need to first create a git repo. This repo will live on the model hub, allowing
|
||||
users to clone it and you (and your organization members) to push to it.
|
||||
|
||||
You can create a model repo directly from the website, `here <https://huggingface.co/new>`.
|
||||
You can create a model repo **directly from `the /new page on the website <https://huggingface.co/new>`__.**
|
||||
|
||||
Alternatively, you can use the ``transformers-cli``. The next steps describe that process:
|
||||
|
||||
@@ -82,12 +82,12 @@ This creates a repo on the model hub, which can be cloned.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
git clone https://huggingface.co/username/your-model-name
|
||||
|
||||
# Make sure you have git-lfs installed
|
||||
# (https://git-lfs.github.com/)
|
||||
git lfs install
|
||||
|
||||
git clone https://huggingface.co/username/your-model-name
|
||||
|
||||
When you have your local clone of your repo and lfs installed, you can then add/remove from that clone as you would
|
||||
with any other git repo.
|
||||
|
||||
@@ -98,8 +98,12 @@ with any other git repo.
|
||||
echo "hello" >> README.md
|
||||
git add . && git commit -m "Update from $USER"
|
||||
|
||||
We are intentionally not wrapping git too much, so as to stay intuitive and easy-to-use.
|
||||
We are intentionally not wrapping git too much, so that you can go on with the workflow you're used to and the tools
|
||||
you already know.
|
||||
|
||||
The only learning curve you might have compared to regular git is the one for git-lfs. The documentation at
|
||||
`git-lfs.github.com <https://git-lfs.github.com/>`__ is decent, but we'll work on a tutorial with some tips and tricks
|
||||
in the coming weeks!
|
||||
|
||||
Make your model work on all frameworks
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
@@ -110,7 +114,7 @@ Make your model work on all frameworks
|
||||
You probably have your favorite framework, but so will other users! That's why it's best to upload your model with both
|
||||
PyTorch `and` TensorFlow checkpoints to make it easier to use (if you skip this step, users will still be able to load
|
||||
your model in another framework, but it will be slower, as it will have to be converted on the fly). Don't worry, it's
|
||||
super easy to do (and in a future version, it will all be automatic). You will need to install both PyTorch and
|
||||
super easy to do (and in a future version, it might all be automatic). You will need to install both PyTorch and
|
||||
TensorFlow for this step, but you don't need to worry about the GPU, so it should be very easy. Check the `TensorFlow
|
||||
installation page <https://www.tensorflow.org/install/pip#tensorflow-2.0-rc-is-available>`__ and/or the `PyTorch
|
||||
installation page <https://pytorch.org/get-started/locally/#start-locally>`__ to see how.
|
||||
@@ -192,7 +196,7 @@ status`` command:
|
||||
git add --all
|
||||
git status
|
||||
|
||||
Finally, the files should be comitted:
|
||||
Finally, the files should be committed:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
@@ -210,23 +214,20 @@ This will upload the folder containing the weights, tokenizer and configuration
|
||||
Add a model card
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
To make sure everyone knows what your model can do, what its limitations and potential bias or ethetical
|
||||
considerations, please add a README.md model card to the 🤗 Transformers repo under `model_cards/`. It should then be
|
||||
placed in a subfolder with your username or organization, then another subfolder named like your model
|
||||
(`awesome-name-you-picked`). Or just click on the "Create a model card on GitHub" button on the model page, it will get
|
||||
you directly to the right location. If you need one, `here <https://github.com/huggingface/model_card>`__ is a model
|
||||
card template (meta-suggestions are welcome).
|
||||
To make sure everyone knows what your model can do, what its limitations, potential bias or ethical considerations are,
|
||||
please add a README.md model card to your model repo. You can just create it, or there's also a convenient button
|
||||
titled "Add a README.md" on your model page. A model card template can be found `here
|
||||
<https://github.com/huggingface/model_card>`__ (meta-suggestions are welcome). model card template (meta-suggestions
|
||||
are welcome).
|
||||
|
||||
.. note::
|
||||
|
||||
Model cards used to live in the 🤗 Transformers repo under `model_cards/`, but for consistency and scalability we
|
||||
migrated every model card from the repo to its corresponding huggingface.co model repo.
|
||||
|
||||
If your model is fine-tuned from another model coming from the model hub (all 🤗 Transformers pretrained models do),
|
||||
don't forget to link to its model card so that people can fully trace how your model was built.
|
||||
|
||||
If you have never made a pull request to the 🤗 Transformers repo, look at the :doc:`contributing guide <contributing>`
|
||||
to see the steps to follow.
|
||||
|
||||
.. note::
|
||||
|
||||
You can also send your model card in the folder you uploaded with the CLI by placing it in a `README.md` file
|
||||
inside `path/to/awesome-name-you-picked/`.
|
||||
|
||||
Using your model
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
@@ -262,7 +263,8 @@ First you need to install `git-lfs` in the environment used by the notebook:
|
||||
|
||||
sudo apt-get install git-lfs
|
||||
|
||||
Then you can use the :obj:`transformers-cli` to create your new repo:
|
||||
Then you can use either create a repo directly from `huggingface.co <https://huggingface.co/>`__ , or use the
|
||||
:obj:`transformers-cli` to create it:
|
||||
|
||||
|
||||
.. code-block:: bash
|
||||
@@ -274,13 +276,14 @@ Once it's created, you can clone it and configure it (replace username by your u
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
git lfs install
|
||||
|
||||
git clone https://username:password@huggingface.co/username/your-model-name
|
||||
# Alternatively if you have a token,
|
||||
# you can use it instead of your password
|
||||
git clone https://username:token@huggingface.co/username/your-model-name
|
||||
|
||||
cd your-model-name
|
||||
git lfs install
|
||||
git config --global user.email "email@example.com"
|
||||
# Tip: using the same email than for your huggingface.co account will link your commits to your profile
|
||||
git config --global user.name "Your name"
|
||||
|
||||
@@ -1142,3 +1142,66 @@ To start a debugger at the point of the warning, do this:
|
||||
.. code-block:: bash
|
||||
|
||||
pytest tests/test_logging.py -W error::UserWarning --pdb
|
||||
|
||||
|
||||
|
||||
Testing Experimental CI Features
|
||||
-----------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
Testing CI features can be potentially problematic as it can interfere with the normal CI functioning. Therefore if a
|
||||
new CI feature is to be added, it should be done as following.
|
||||
|
||||
1. Create a new dedicated job that tests what needs to be tested
|
||||
2. The new job must always succeed so that it gives us a green ✓ (details below).
|
||||
3. Let it run for some days to see that a variety of different PR types get to run on it (user fork branches,
|
||||
non-forked branches, branches originating from github.com UI direct file edit, various forced pushes, etc. - there
|
||||
are so many) while monitoring the experimental job's logs (not the overall job green as it's purposefully always
|
||||
green)
|
||||
4. When it's clear that everything is solid, then merge the new changes into existing jobs.
|
||||
|
||||
That way experiments on CI functionality itself won't interfere with the normal workflow.
|
||||
|
||||
Now how can we make the job always succeed while the new CI feature is being developed?
|
||||
|
||||
Some CIs, like TravisCI support ignore-step-failure and will report the overall job as successful, but CircleCI and
|
||||
Github Actions as of this writing don't support that.
|
||||
|
||||
So the following workaround can be used:
|
||||
|
||||
1. ``set +euo pipefail`` at the beginning of the run command to suppress most potential failures in the bash script.
|
||||
2. the last command must be a success: ``echo "done"`` or just ``true`` will do
|
||||
|
||||
Here is an example:
|
||||
|
||||
.. code-block:: yaml
|
||||
|
||||
- run:
|
||||
name: run CI experiment
|
||||
command: |
|
||||
set +euo pipefail
|
||||
echo "setting run-all-despite-any-errors-mode"
|
||||
this_command_will_fail
|
||||
echo "but bash continues to run"
|
||||
# emulate another failure
|
||||
false
|
||||
# but the last command must be a success
|
||||
echo "during experiment do not remove: reporting success to CI, even if there were failures"
|
||||
|
||||
For simple commands you could also do:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
cmd_that_may_fail || true
|
||||
|
||||
Of course, once satisfied with the results, integrate the experimental step or job with the rest of the normal jobs,
|
||||
while removing ``set +euo pipefail`` or any other things you may have added to ensure that the experimental job doesn't
|
||||
interfere with the normal CI functioning.
|
||||
|
||||
This whole process would have been much easier if we only could set something like ``allow-failure`` for the
|
||||
experimental step, and let it fail without impacting the overall status of PRs. But as mentioned earlier CircleCI and
|
||||
Github Actions don't support it at the moment.
|
||||
|
||||
You can vote for this feature and see where it is at at these CI-specific threads:
|
||||
|
||||
* `Github Actions: <https://github.com/actions/toolkit/issues/399>`__
|
||||
* `CircleCI: <https://ideas.circleci.com/ideas/CCI-I-344>`__
|
||||
|
||||
+2
-2
@@ -55,11 +55,11 @@ Coming soon!
|
||||
|---|---|:---:|:---:|:---:|:---:|
|
||||
| [**`language-modeling`**](https://github.com/huggingface/transformers/tree/master/examples/language-modeling) | Raw text | ✅ | - | ✅ | [](https://colab.research.google.com/github/huggingface/blog/blob/master/notebooks/01_how_to_train.ipynb)
|
||||
| [**`multiple-choice`**](https://github.com/huggingface/transformers/tree/master/examples/multiple-choice) | SWAG, RACE, ARC | ✅ | ✅ | - | [](https://colab.research.google.com/github/ViktorAlm/notebooks/blob/master/MPC_GPU_Demo_for_TF_and_PT.ipynb)
|
||||
| [**`question-answering`**](https://github.com/huggingface/transformers/tree/master/examples/question-answering) | SQuAD | ✅ | ✅ | ✅ | -
|
||||
| [**`question-answering`**](https://github.com/huggingface/transformers/tree/master/examples/question-answering) | SQuAD | ✅ | ✅ | ✅ | [](https://github.com/huggingface/notebooks/blob/master/examples/question_answering.ipynb)
|
||||
| [**`summarization`**](https://github.com/huggingface/transformers/tree/master/examples/seq2seq) | CNN/Daily Mail | ✅ | - | - | -
|
||||
| [**`text-classification`**](https://github.com/huggingface/transformers/tree/master/examples/text-classification) | GLUE, XNLI | ✅ | ✅ | ✅ | [](https://github.com/huggingface/notebooks/blob/master/examples/text_classification.ipynb)
|
||||
| [**`text-generation`**](https://github.com/huggingface/transformers/tree/master/examples/text-generation) | - | n/a | n/a | - | [](https://colab.research.google.com/github/huggingface/blog/blob/master/notebooks/02_how_to_generate.ipynb)
|
||||
| [**`token-classification`**](https://github.com/huggingface/transformers/tree/master/examples/token-classification) | CoNLL NER | ✅ | ✅ | ✅ | -
|
||||
| [**`token-classification`**](https://github.com/huggingface/transformers/tree/master/examples/token-classification) | CoNLL NER | ✅ | ✅ | ✅ | [](https://github.com/huggingface/notebooks/blob/master/examples/token_classification.ipynb)
|
||||
| [**`translation`**](https://github.com/huggingface/transformers/tree/master/examples/seq2seq) | WMT | ✅ | - | - | -
|
||||
|
||||
|
||||
|
||||
@@ -113,6 +113,12 @@ class DataTrainingArguments:
|
||||
overwrite_cache: bool = field(
|
||||
default=False, metadata={"help": "Overwrite the cached training and evaluation sets"}
|
||||
)
|
||||
validation_split_percentage: Optional[int] = field(
|
||||
default=5,
|
||||
metadata={
|
||||
"help": "The percentage of the train set used as validation set in case there's no validation split"
|
||||
},
|
||||
)
|
||||
preprocessing_num_workers: Optional[int] = field(
|
||||
default=None,
|
||||
metadata={"help": "The number of processes to use for the preprocessing."},
|
||||
@@ -188,6 +194,17 @@ def main():
|
||||
if data_args.dataset_name is not None:
|
||||
# Downloading and loading a dataset from the hub.
|
||||
datasets = load_dataset(data_args.dataset_name, data_args.dataset_config_name)
|
||||
if "validation" not in datasets.keys():
|
||||
datasets["validation"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[:{data_args.validation_split_percentage}%]",
|
||||
)
|
||||
datasets["train"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[{data_args.validation_split_percentage}%:]",
|
||||
)
|
||||
else:
|
||||
data_files = {}
|
||||
if data_args.train_file is not None:
|
||||
|
||||
@@ -103,6 +103,12 @@ class DataTrainingArguments:
|
||||
overwrite_cache: bool = field(
|
||||
default=False, metadata={"help": "Overwrite the cached training and evaluation sets"}
|
||||
)
|
||||
validation_split_percentage: Optional[int] = field(
|
||||
default=5,
|
||||
metadata={
|
||||
"help": "The percentage of the train set used as validation set in case there's no validation split"
|
||||
},
|
||||
)
|
||||
max_seq_length: Optional[int] = field(
|
||||
default=None,
|
||||
metadata={
|
||||
@@ -199,6 +205,17 @@ def main():
|
||||
if data_args.dataset_name is not None:
|
||||
# Downloading and loading a dataset from the hub.
|
||||
datasets = load_dataset(data_args.dataset_name, data_args.dataset_config_name)
|
||||
if "validation" not in datasets.keys():
|
||||
datasets["validation"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[:{data_args.validation_split_percentage}%]",
|
||||
)
|
||||
datasets["train"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[{data_args.validation_split_percentage}%:]",
|
||||
)
|
||||
else:
|
||||
data_files = {}
|
||||
if data_args.train_file is not None:
|
||||
|
||||
@@ -134,6 +134,12 @@ class DataTrainingArguments:
|
||||
overwrite_cache: bool = field(
|
||||
default=False, metadata={"help": "Overwrite the cached training and evaluation sets"}
|
||||
)
|
||||
validation_split_percentage: Optional[int] = field(
|
||||
default=5,
|
||||
metadata={
|
||||
"help": "The percentage of the train set used as validation set in case there's no validation split"
|
||||
},
|
||||
)
|
||||
max_seq_length: Optional[int] = field(
|
||||
default=None,
|
||||
metadata={
|
||||
@@ -379,7 +385,7 @@ def training_step(optimizer, batch, dropout_rng):
|
||||
# Hide away tokens which doesn't participate in the optimization
|
||||
token_mask = jnp.where(targets > 0, 1.0, 0.0)
|
||||
|
||||
pooled, logits = model(**batch, params=params, dropout_rng=dropout_rng, train=True)
|
||||
logits = model(**batch, params=params, dropout_rng=dropout_rng, train=True)[0]
|
||||
loss, weight_sum = cross_entropy(logits, targets, token_mask)
|
||||
return loss / weight_sum
|
||||
|
||||
@@ -401,7 +407,7 @@ def eval_step(params, batch):
|
||||
|
||||
# Hide away tokens which doesn't participate in the optimization
|
||||
token_mask = jnp.where(targets > 0, 1.0, 0.0)
|
||||
_, logits = model(**batch, params=params, train=False)
|
||||
logits = model(**batch, params=params, train=False)[0]
|
||||
|
||||
return compute_metrics(logits, targets, token_mask)
|
||||
|
||||
@@ -413,7 +419,7 @@ def generate_batch_splits(samples_idx: jnp.ndarray, batch_size: int) -> jnp.ndar
|
||||
if samples_to_remove != 0:
|
||||
samples_idx = samples_idx[:-samples_to_remove]
|
||||
sections_split = nb_samples // batch_size
|
||||
batch_idx = jnp.split(samples_idx, sections_split)
|
||||
batch_idx = np.split(samples_idx, sections_split)
|
||||
return batch_idx
|
||||
|
||||
|
||||
@@ -473,6 +479,17 @@ if __name__ == "__main__":
|
||||
if data_args.dataset_name is not None:
|
||||
# Downloading and loading a dataset from the hub.
|
||||
datasets = load_dataset(data_args.dataset_name, data_args.dataset_config_name)
|
||||
if "validation" not in datasets.keys():
|
||||
datasets["validation"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[:{data_args.validation_split_percentage}%]",
|
||||
)
|
||||
datasets["train"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[{data_args.validation_split_percentage}%:]",
|
||||
)
|
||||
else:
|
||||
data_files = {}
|
||||
if data_args.train_file is not None:
|
||||
@@ -525,9 +542,9 @@ if __name__ == "__main__":
|
||||
|
||||
def tokenize_function(examples):
|
||||
# Remove empty lines
|
||||
examples["text"] = [line for line in examples["text"] if len(line) > 0 and not line.isspace()]
|
||||
examples = [line for line in examples if len(line) > 0 and not line.isspace()]
|
||||
return tokenizer(
|
||||
examples["text"],
|
||||
examples,
|
||||
return_special_tokens_mask=True,
|
||||
padding=padding,
|
||||
truncation=True,
|
||||
@@ -536,9 +553,10 @@ if __name__ == "__main__":
|
||||
|
||||
tokenized_datasets = datasets.map(
|
||||
tokenize_function,
|
||||
input_columns=[text_column_name],
|
||||
batched=True,
|
||||
num_proc=data_args.preprocessing_num_workers,
|
||||
remove_columns=[text_column_name],
|
||||
remove_columns=column_names,
|
||||
load_from_cache_file=not data_args.overwrite_cache,
|
||||
)
|
||||
|
||||
@@ -554,8 +572,13 @@ if __name__ == "__main__":
|
||||
rng = jax.random.PRNGKey(training_args.seed)
|
||||
dropout_rngs = jax.random.split(rng, jax.local_device_count())
|
||||
|
||||
model = FlaxBertForMaskedLM.from_pretrained("bert-base-cased", dtype=jnp.float32, dropout_rate=0.1)
|
||||
model.init(jax.random.PRNGKey(training_args.seed), (training_args.train_batch_size, model.config.max_length))
|
||||
model = FlaxBertForMaskedLM.from_pretrained(
|
||||
"bert-base-cased",
|
||||
dtype=jnp.float32,
|
||||
input_shape=(training_args.train_batch_size, config.max_position_embeddings),
|
||||
seed=training_args.seed,
|
||||
dropout_rate=0.1,
|
||||
)
|
||||
|
||||
# Setup optimizer
|
||||
optimizer = Adam(
|
||||
@@ -566,8 +589,9 @@ if __name__ == "__main__":
|
||||
).create(model.params)
|
||||
|
||||
# Create learning rate scheduler
|
||||
# warmup_steps = 0 causes the Flax optimizer to return NaNs; warmup_steps = 1 is functionally equivalent.
|
||||
lr_scheduler_fn = create_learning_rate_scheduler(
|
||||
base_learning_rate=training_args.learning_rate, warmup_steps=training_args.warmup_steps
|
||||
base_learning_rate=training_args.learning_rate, warmup_steps=min(training_args.warmup_steps, 1)
|
||||
)
|
||||
|
||||
# Create parallel version of the training and evaluation steps
|
||||
@@ -606,13 +630,13 @@ if __name__ == "__main__":
|
||||
epochs.write(f"Loss: {loss}")
|
||||
|
||||
# ======================== Evaluating ==============================
|
||||
nb_eval_samples = len(tokenized_datasets["test"])
|
||||
nb_eval_samples = len(tokenized_datasets["validation"])
|
||||
eval_samples_idx = jnp.arange(nb_eval_samples)
|
||||
eval_batch_idx = generate_batch_splits(eval_samples_idx, eval_batch_size)
|
||||
|
||||
eval_metrics = []
|
||||
for i, batch_idx in enumerate(tqdm(eval_batch_idx, desc="Evaluating ...", position=2)):
|
||||
samples = [tokenized_datasets["test"][int(idx)] for idx in batch_idx]
|
||||
samples = [tokenized_datasets["validation"][int(idx)] for idx in batch_idx]
|
||||
model_inputs = data_collator(samples, pad_to_multiple_of=16)
|
||||
|
||||
# Model forward
|
||||
|
||||
@@ -91,6 +91,12 @@ class DataTrainingArguments:
|
||||
Arguments pertaining to what data we are going to input our model for training and eval.
|
||||
"""
|
||||
|
||||
dataset_name: Optional[str] = field(
|
||||
default=None, metadata={"help": "The name of the dataset to use (via the datasets library)."}
|
||||
)
|
||||
dataset_config_name: Optional[str] = field(
|
||||
default=None, metadata={"help": "The configuration name of the dataset to use (via the datasets library)."}
|
||||
)
|
||||
train_file: Optional[str] = field(default=None, metadata={"help": "The input training data file (a text file)."})
|
||||
validation_file: Optional[str] = field(
|
||||
default=None,
|
||||
@@ -107,6 +113,12 @@ class DataTrainingArguments:
|
||||
overwrite_cache: bool = field(
|
||||
default=False, metadata={"help": "Overwrite the cached training and evaluation sets"}
|
||||
)
|
||||
validation_split_percentage: Optional[int] = field(
|
||||
default=5,
|
||||
metadata={
|
||||
"help": "The percentage of the train set used as validation set in case there's no validation split"
|
||||
},
|
||||
)
|
||||
max_seq_length: Optional[int] = field(
|
||||
default=None,
|
||||
metadata={
|
||||
@@ -203,15 +215,30 @@ def main():
|
||||
#
|
||||
# In distributed training, the load_dataset function guarantee that only one local process can concurrently
|
||||
# download the dataset.
|
||||
data_files = {}
|
||||
if data_args.train_file is not None:
|
||||
data_files["train"] = data_args.train_file
|
||||
if data_args.validation_file is not None:
|
||||
data_files["validation"] = data_args.validation_file
|
||||
extension = data_args.train_file.split(".")[-1]
|
||||
if extension == "txt":
|
||||
extension = "text"
|
||||
datasets = load_dataset(extension, data_files=data_files)
|
||||
if data_args.dataset_name is not None:
|
||||
# Downloading and loading a dataset from the hub.
|
||||
datasets = load_dataset(data_args.dataset_name, data_args.dataset_config_name)
|
||||
if "validation" not in datasets.keys():
|
||||
datasets["validation"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[:{data_args.validation_split_percentage}%]",
|
||||
)
|
||||
datasets["train"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[{data_args.validation_split_percentage}%:]",
|
||||
)
|
||||
else:
|
||||
data_files = {}
|
||||
if data_args.train_file is not None:
|
||||
data_files["train"] = data_args.train_file
|
||||
if data_args.validation_file is not None:
|
||||
data_files["validation"] = data_args.validation_file
|
||||
extension = data_args.train_file.split(".")[-1]
|
||||
if extension == "txt":
|
||||
extension = "text"
|
||||
datasets = load_dataset(extension, data_files=data_files)
|
||||
# See more about loading any type of standard or custom dataset (from files, python dict, pandas DataFrame, etc) at
|
||||
# https://huggingface.co/docs/datasets/loading_datasets.html.
|
||||
|
||||
|
||||
@@ -93,6 +93,12 @@ class DataTrainingArguments:
|
||||
overwrite_cache: bool = field(
|
||||
default=False, metadata={"help": "Overwrite the cached training and evaluation sets"}
|
||||
)
|
||||
validation_split_percentage: Optional[int] = field(
|
||||
default=5,
|
||||
metadata={
|
||||
"help": "The percentage of the train set used as validation set in case there's no validation split"
|
||||
},
|
||||
)
|
||||
max_seq_length: int = field(
|
||||
default=512,
|
||||
metadata={
|
||||
@@ -196,6 +202,17 @@ def main():
|
||||
if data_args.dataset_name is not None:
|
||||
# Downloading and loading a dataset from the hub.
|
||||
datasets = load_dataset(data_args.dataset_name, data_args.dataset_config_name)
|
||||
if "validation" not in datasets.keys():
|
||||
datasets["validation"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[:{data_args.validation_split_percentage}%]",
|
||||
)
|
||||
datasets["train"] = load_dataset(
|
||||
data_args.dataset_name,
|
||||
data_args.dataset_config_name,
|
||||
split=f"train[{data_args.validation_split_percentage}%:]",
|
||||
)
|
||||
else:
|
||||
data_files = {}
|
||||
if data_args.train_file is not None:
|
||||
|
||||
@@ -76,9 +76,7 @@ def postprocess_qa_predictions(
|
||||
assert len(predictions) == 2, "`predictions` should be a tuple with two elements (start_logits, end_logits)."
|
||||
all_start_logits, all_end_logits = predictions
|
||||
|
||||
assert len(predictions[0]) == len(
|
||||
features
|
||||
), f"Got {len(predictions[0])} predicitions and {len(features)} features."
|
||||
assert len(predictions[0]) == len(features), f"Got {len(predictions[0])} predictions and {len(features)} features."
|
||||
|
||||
# Build a map example to its corresponding features.
|
||||
example_id_to_index = {k: i for i, k in enumerate(examples["id"])}
|
||||
@@ -118,7 +116,7 @@ def postprocess_qa_predictions(
|
||||
|
||||
# Update minimum null prediction.
|
||||
feature_null_score = start_logits[0] + end_logits[0]
|
||||
if min_null_prediction is None or min_null_prediction["score"] < feature_null_score:
|
||||
if min_null_prediction is None or min_null_prediction["score"] > feature_null_score:
|
||||
min_null_prediction = {
|
||||
"offsets": (0, 0),
|
||||
"score": feature_null_score,
|
||||
|
||||
@@ -96,7 +96,7 @@ def evaluate_batch_retrieval(args, rag_model, questions):
|
||||
)["input_ids"].to(args.device)
|
||||
|
||||
question_enc_outputs = rag_model.rag.question_encoder(retriever_input_ids)
|
||||
question_enc_pool_output = question_enc_outputs.pooler_output
|
||||
question_enc_pool_output = question_enc_outputs[0]
|
||||
|
||||
result = rag_model.retriever(
|
||||
retriever_input_ids,
|
||||
|
||||
@@ -97,7 +97,7 @@ The `.source` files are the input, the `.target` files are the desired output.
|
||||
|
||||
### Potential issues
|
||||
|
||||
- native AMP (`--fp16` and no apex) may lead to a huge memory leak and require 10x gpu memory. This has been fixed in pytorch-nightly and the minimal official version to have this fix will be pytorch-1.8. Until then if you have to use mixed precision please use AMP only with pytorch-nightly or NVIDIA's apex. Reference: https://github.com/huggingface/transformers/issues/8403
|
||||
- native AMP (`--fp16` and no apex) may lead to a huge memory leak and require 10x gpu memory. This has been fixed in pytorch-nightly and the minimal official version to have this fix will be pytorch-1.7.1. Until then if you have to use mixed precision please use AMP only with pytorch-nightly or NVIDIA's apex. Reference: https://github.com/huggingface/transformers/issues/8403
|
||||
|
||||
|
||||
### Tips and Tricks
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
@@ -119,6 +120,46 @@ class DataTrainingArguments:
|
||||
)
|
||||
|
||||
|
||||
def speed_metrics(split, start_time, num_samples):
|
||||
"""
|
||||
Measure and return speed performance metrics.
|
||||
|
||||
This function requires a time snapshot `start_time` before the operation to be measured starts and this
|
||||
function should be run immediately after the operation to be measured has completed.
|
||||
|
||||
Args:
|
||||
- split: one of train, val, test
|
||||
- start_time: operation start time
|
||||
- num_samples: number of samples processed
|
||||
|
||||
"""
|
||||
runtime = time.time() - start_time
|
||||
result = {}
|
||||
|
||||
samples_per_second = 1 / (runtime / num_samples)
|
||||
result[f"{split}_samples_per_second"] = round(samples_per_second, 3)
|
||||
result[f"{split}_runtime"] = round(runtime, 4)
|
||||
|
||||
result[f"{split}_n_ojbs"] = num_samples
|
||||
return result
|
||||
|
||||
|
||||
def handle_metrics(split, metrics, output_dir):
|
||||
"""
|
||||
Log and save metrics
|
||||
|
||||
Args:
|
||||
- split: one of train, val, test
|
||||
- metrics: metrics dict
|
||||
- output_dir: where to save the metrics
|
||||
"""
|
||||
|
||||
logger.info(f"***** {split} metrics *****")
|
||||
for key, value in metrics.items():
|
||||
logger.info(f" {key} = {value}")
|
||||
save_json(metrics, os.path.join(output_dir, f"{split}_results.json"))
|
||||
|
||||
|
||||
def main():
|
||||
# See all possible arguments in src/transformers/training_args.py
|
||||
# or by passing the --help flag to this script.
|
||||
@@ -265,45 +306,56 @@ def main():
|
||||
data_args=data_args,
|
||||
)
|
||||
|
||||
all_metrics = {}
|
||||
# Training
|
||||
if training_args.do_train:
|
||||
logger.info("*** Train ***")
|
||||
|
||||
start_time = time.time()
|
||||
trainer.train(
|
||||
model_path=model_args.model_name_or_path if os.path.isdir(model_args.model_name_or_path) else None
|
||||
)
|
||||
trainer.save_model()
|
||||
# For convenience, we also re-save the tokenizer to the same directory,
|
||||
# so that you can share your model easily on huggingface.co/models =)
|
||||
metrics = speed_metrics("train", start_time, data_args.n_train)
|
||||
|
||||
trainer.save_model() # this also saves the tokenizer
|
||||
|
||||
if trainer.is_world_process_zero():
|
||||
handle_metrics("train", metrics, training_args.output_dir)
|
||||
all_metrics.update(metrics)
|
||||
|
||||
# Need to save the state, since Trainer.save_model saves only the tokenizer with the model
|
||||
trainer.state.save_to_json(os.path.join(training_args.output_dir, "trainer_state.json"))
|
||||
|
||||
# For convenience, we also re-save the tokenizer to the same directory,
|
||||
# so that you can share your model easily on huggingface.co/models =)
|
||||
tokenizer.save_pretrained(training_args.output_dir)
|
||||
|
||||
# Evaluation
|
||||
eval_results = {}
|
||||
if training_args.do_eval:
|
||||
logger.info("*** Evaluate ***")
|
||||
|
||||
result = trainer.evaluate()
|
||||
start_time = time.time()
|
||||
metrics = trainer.evaluate(metric_key_prefix="val")
|
||||
metrics.update(speed_metrics("val", start_time, data_args.n_val))
|
||||
metrics["val_loss"] = round(metrics["val_loss"], 4)
|
||||
|
||||
if trainer.is_world_process_zero():
|
||||
logger.info("***** Eval results *****")
|
||||
for key, value in result.items():
|
||||
logger.info(" %s = %s", key, value)
|
||||
save_json(result, os.path.join(training_args.output_dir, "eval_results.json"))
|
||||
eval_results.update(result)
|
||||
|
||||
handle_metrics("val", metrics, training_args.output_dir)
|
||||
all_metrics.update(metrics)
|
||||
|
||||
if training_args.do_predict:
|
||||
logging.info("*** Test ***")
|
||||
logger.info("*** Predict ***")
|
||||
|
||||
test_output = trainer.predict(test_dataset=test_dataset)
|
||||
test_metrics = {k.replace("eval", "test"): v for k, v in test_output.metrics.items()}
|
||||
start_time = time.time()
|
||||
test_output = trainer.predict(test_dataset=test_dataset, metric_key_prefix="test")
|
||||
metrics = test_output.metrics
|
||||
metrics.update(speed_metrics("test", start_time, data_args.n_test))
|
||||
|
||||
if trainer.is_world_process_zero():
|
||||
logger.info("***** Test results *****")
|
||||
for key, value in test_metrics.items():
|
||||
logger.info(" %s = %s", key, value)
|
||||
|
||||
save_json(test_metrics, os.path.join(training_args.output_dir, "test_results.json"))
|
||||
eval_results.update(test_metrics)
|
||||
metrics["test_loss"] = round(metrics["test_loss"], 4)
|
||||
handle_metrics("test", metrics, training_args.output_dir)
|
||||
all_metrics.update(metrics)
|
||||
|
||||
if training_args.predict_with_generate:
|
||||
test_preds = tokenizer.batch_decode(
|
||||
@@ -313,8 +365,9 @@ def main():
|
||||
write_txt_file(test_preds, os.path.join(training_args.output_dir, "test_generations.txt"))
|
||||
|
||||
if trainer.is_world_process_zero():
|
||||
save_json(eval_results, "all_results.json")
|
||||
return eval_results
|
||||
save_json(all_metrics, os.path.join(training_args.output_dir, "all_results.json"))
|
||||
|
||||
return all_metrics
|
||||
|
||||
|
||||
def _mp_fn(index):
|
||||
|
||||
@@ -20,6 +20,7 @@ from torch.utils.data import DistributedSampler, RandomSampler
|
||||
|
||||
from transformers import PreTrainedModel, Trainer, logging
|
||||
from transformers.file_utils import is_torch_tpu_available
|
||||
from transformers.integrations import is_fairscale_available
|
||||
from transformers.models.fsmt.configuration_fsmt import FSMTConfig
|
||||
from transformers.optimization import (
|
||||
Adafactor,
|
||||
@@ -35,6 +36,10 @@ from transformers.trainer_pt_utils import get_tpu_sampler
|
||||
from transformers.training_args import ParallelMode
|
||||
|
||||
|
||||
if is_fairscale_available():
|
||||
from fairscale.optim import OSS
|
||||
|
||||
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
arg_to_scheduler = {
|
||||
@@ -99,18 +104,25 @@ class Seq2SeqTrainer(Trainer):
|
||||
"weight_decay": 0.0,
|
||||
},
|
||||
]
|
||||
optimizer_cls = Adafactor if self.args.adafactor else AdamW
|
||||
if self.args.adafactor:
|
||||
self.optimizer = Adafactor(
|
||||
optimizer_grouped_parameters,
|
||||
lr=self.args.learning_rate,
|
||||
scale_parameter=False,
|
||||
relative_step=False,
|
||||
)
|
||||
|
||||
optimizer_cls = Adafactor
|
||||
optimizer_kwargs = {"scale_parameter": False, "relative_step": False}
|
||||
else:
|
||||
self.optimizer = AdamW(
|
||||
optimizer_grouped_parameters, lr=self.args.learning_rate, eps=self.args.adam_epsilon
|
||||
optimizer_cls = AdamW
|
||||
optimizer_kwargs = {
|
||||
"betas": (self.args.adam_beta1, self.args.adam_beta2),
|
||||
"eps": self.args.adam_epsilon,
|
||||
}
|
||||
optimizer_kwargs["lr"] = self.args.learning_rate
|
||||
if self.sharded_dpp:
|
||||
self.optimizer = OSS(
|
||||
params=optimizer_grouped_parameters,
|
||||
optim=optimizer_cls,
|
||||
**optimizer_kwargs,
|
||||
)
|
||||
else:
|
||||
self.optimizer = optimizer_cls(optimizer_grouped_parameters, **optimizer_kwargs)
|
||||
|
||||
if self.lr_scheduler is None:
|
||||
self.lr_scheduler = self._get_lr_scheduler(num_training_steps)
|
||||
|
||||
@@ -462,7 +462,7 @@ def save_git_info(folder_path: str) -> None:
|
||||
|
||||
def save_json(content, path, indent=4, **json_dump_kwargs):
|
||||
with open(path, "w") as f:
|
||||
json.dump(content, f, indent=indent, **json_dump_kwargs)
|
||||
json.dump(content, f, indent=indent, sort_keys=True, **json_dump_kwargs)
|
||||
|
||||
|
||||
def load_json(path):
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
---
|
||||
language: ja
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
## Japanese ELECTRA-small
|
||||
|
||||
We provide a Japanese **ELECTRA-Small** model, as described in [ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators](https://openreview.net/pdf?id=r1xMH1BtvB).
|
||||
|
||||
Our pretraining process employs subword units derived from the [Japanese Wikipedia](https://dumps.wikimedia.org/jawiki/latest), using the [Byte-Pair Encoding](https://www.aclweb.org/anthology/P16-1162.pdf) method and building on an initial tokenization with [mecab-ipadic-NEologd](https://github.com/neologd/mecab-ipadic-neologd). For optimal performance, please take care to set your MeCab dictionary appropriately.
|
||||
|
||||
## How to use the discriminator in `transformers`
|
||||
|
||||
```
|
||||
from transformers import BertJapaneseTokenizer, ElectraForPreTraining
|
||||
|
||||
tokenizer = BertJapaneseTokenizer.from_pretrained('Cinnamon/electra-small-japanese-discriminator', mecab_kwargs={"mecab_option": "-d /usr/lib/x86_64-linux-gnu/mecab/dic/mecab-ipadic-neologd"})
|
||||
|
||||
model = ElectraForPreTraining.from_pretrained('Cinnamon/electra-small-japanese-discriminator')
|
||||
```
|
||||
@@ -1,18 +0,0 @@
|
||||
---
|
||||
language: ja
|
||||
---
|
||||
## Japanese ELECTRA-small
|
||||
|
||||
We provide a Japanese **ELECTRA-Small** model, as described in [ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators](https://openreview.net/pdf?id=r1xMH1BtvB).
|
||||
|
||||
Our pretraining process employs subword units derived from the [Japanese Wikipedia](https://dumps.wikimedia.org/jawiki/latest), using the [Byte-Pair Encoding](https://www.aclweb.org/anthology/P16-1162.pdf) method and building on an initial tokenization with [mecab-ipadic-NEologd](https://github.com/neologd/mecab-ipadic-neologd). For optimal performance, please take care to set your MeCab dictionary appropriately.
|
||||
|
||||
```
|
||||
# ELECTRA-small generator usage
|
||||
|
||||
from transformers import BertJapaneseTokenizer, ElectraForMaskedLM
|
||||
|
||||
tokenizer = BertJapaneseTokenizer.from_pretrained('Cinnamon/electra-small-japanese-generator', mecab_kwargs={"mecab_option": "-d /usr/lib/x86_64-linux-gnu/mecab/dic/mecab-ipadic-neologd"})
|
||||
|
||||
model = ElectraForMaskedLM.from_pretrained('Cinnamon/electra-small-japanese-generator')
|
||||
```
|
||||
@@ -1,142 +0,0 @@
|
||||
---
|
||||
language: da
|
||||
tags:
|
||||
- bert
|
||||
- masked-lm
|
||||
license: cc-by-4.0
|
||||
datasets:
|
||||
- common_crawl
|
||||
- wikipedia
|
||||
pipeline_tag: fill-mask
|
||||
widget:
|
||||
- text: "København er [MASK] i Danmark."
|
||||
---
|
||||
|
||||
# Danish BERT (uncased) model
|
||||
|
||||
[BotXO.ai](https://www.botxo.ai/) developed this model. For data and training details see their [GitHub repository](https://github.com/botxo/nordic_bert).
|
||||
|
||||
The original model was trained in TensorFlow then I converted it to Pytorch using [transformers-cli](https://huggingface.co/transformers/converting_tensorflow_models.html?highlight=cli).
|
||||
|
||||
For TensorFlow version download here: https://www.dropbox.com/s/19cjaoqvv2jicq9/danish_bert_uncased_v2.zip?dl=1
|
||||
|
||||
|
||||
## Architecture
|
||||
|
||||
```python
|
||||
from transformers import AutoModelForPreTraining
|
||||
|
||||
model = AutoModelForPreTraining.from_pretrained("DJSammy/bert-base-danish-uncased_BotXO,ai")
|
||||
|
||||
params = list(model.named_parameters())
|
||||
print('danish_bert_uncased_v2 has {:} different named parameters.\n'.format(len(params)))
|
||||
|
||||
print('==== Embedding Layer ====\n')
|
||||
for p in params[0:5]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== First Transformer ====\n')
|
||||
for p in params[5:21]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== Last Transformer ====\n')
|
||||
for p in params[181:197]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== Output Layer ====\n')
|
||||
for p in params[197:]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
# danish_bert_uncased_v2 has 206 different named parameters.
|
||||
|
||||
# ==== Embedding Layer ====
|
||||
|
||||
# bert.embeddings.word_embeddings.weight (32000, 768)
|
||||
# bert.embeddings.position_embeddings.weight (512, 768)
|
||||
# bert.embeddings.token_type_embeddings.weight (2, 768)
|
||||
# bert.embeddings.LayerNorm.weight (768,)
|
||||
# bert.embeddings.LayerNorm.bias (768,)
|
||||
|
||||
# ==== First Transformer ====
|
||||
|
||||
# bert.encoder.layer.0.attention.self.query.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.query.bias (768,)
|
||||
# bert.encoder.layer.0.attention.self.key.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.key.bias (768,)
|
||||
# bert.encoder.layer.0.attention.self.value.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.value.bias (768,)
|
||||
# bert.encoder.layer.0.attention.output.dense.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.output.dense.bias (768,)
|
||||
# bert.encoder.layer.0.attention.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.0.attention.output.LayerNorm.bias (768,)
|
||||
# bert.encoder.layer.0.intermediate.dense.weight (3072, 768)
|
||||
# bert.encoder.layer.0.intermediate.dense.bias (3072,)
|
||||
# bert.encoder.layer.0.output.dense.weight (768, 3072)
|
||||
# bert.encoder.layer.0.output.dense.bias (768,)
|
||||
# bert.encoder.layer.0.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.0.output.LayerNorm.bias (768,)
|
||||
|
||||
# ==== Last Transformer ====
|
||||
|
||||
# bert.encoder.layer.11.attention.self.query.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.query.bias (768,)
|
||||
# bert.encoder.layer.11.attention.self.key.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.key.bias (768,)
|
||||
# bert.encoder.layer.11.attention.self.value.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.value.bias (768,)
|
||||
# bert.encoder.layer.11.attention.output.dense.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.output.dense.bias (768,)
|
||||
# bert.encoder.layer.11.attention.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.11.attention.output.LayerNorm.bias (768,)
|
||||
# bert.encoder.layer.11.intermediate.dense.weight (3072, 768)
|
||||
# bert.encoder.layer.11.intermediate.dense.bias (3072,)
|
||||
# bert.encoder.layer.11.output.dense.weight (768, 3072)
|
||||
# bert.encoder.layer.11.output.dense.bias (768,)
|
||||
# bert.encoder.layer.11.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.11.output.LayerNorm.bias (768,)
|
||||
|
||||
# ==== Output Layer ====
|
||||
|
||||
# bert.pooler.dense.weight (768, 768)
|
||||
# bert.pooler.dense.bias (768,)
|
||||
# cls.predictions.bias (32000,)
|
||||
# cls.predictions.transform.dense.weight (768, 768)
|
||||
# cls.predictions.transform.dense.bias (768,)
|
||||
# cls.predictions.transform.LayerNorm.weight (768,)
|
||||
# cls.predictions.transform.LayerNorm.bias (768,)
|
||||
# cls.seq_relationship.weight (2, 768)
|
||||
# cls.seq_relationship.bias (2,)
|
||||
```
|
||||
|
||||
## Example Pipeline
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
unmasker = pipeline('fill-mask', model='DJSammy/bert-base-danish-uncased_BotXO,ai')
|
||||
|
||||
unmasker('København er [MASK] i Danmark.')
|
||||
|
||||
# Copenhagen is the [MASK] of Denmark.
|
||||
# =>
|
||||
|
||||
# [{'score': 0.788068950176239,
|
||||
# 'sequence': '[CLS] københavn er hovedstad i danmark. [SEP]',
|
||||
# 'token': 12610,
|
||||
# 'token_str': 'hovedstad'},
|
||||
# {'score': 0.07606703042984009,
|
||||
# 'sequence': '[CLS] københavn er hovedstaden i danmark. [SEP]',
|
||||
# 'token': 8108,
|
||||
# 'token_str': 'hovedstaden'},
|
||||
# {'score': 0.04299738258123398,
|
||||
# 'sequence': '[CLS] københavn er metropol i danmark. [SEP]',
|
||||
# 'token': 23305,
|
||||
# 'token_str': 'metropol'},
|
||||
# {'score': 0.008163209073245525,
|
||||
# 'sequence': '[CLS] københavn er ikke i danmark. [SEP]',
|
||||
# 'token': 89,
|
||||
# 'token_str': 'ikke'},
|
||||
# {'score': 0.006238455418497324,
|
||||
# 'sequence': '[CLS] københavn er ogsa i danmark. [SEP]',
|
||||
# 'token': 25253,
|
||||
# 'token_str': 'ogsa'}]
|
||||
```
|
||||
@@ -1,14 +0,0 @@
|
||||
---
|
||||
language:
|
||||
- bg
|
||||
- cs
|
||||
- pl
|
||||
- ru
|
||||
---
|
||||
|
||||
# bert-base-bg-cs-pl-ru-cased
|
||||
|
||||
SlavicBERT\[1\] \(Slavic \(bg, cs, pl, ru\), cased, 12‑layer, 768‑hidden, 12‑heads, 180M parameters\) was trained on Russian News and four Wikipedias: Bulgarian, Czech, Polish, and Russian. Subtoken vocabulary was built using this data. Multilingual BERT was used as an initialization for SlavicBERT.
|
||||
|
||||
|
||||
\[1\]: Arkhipov M., Trofimova M., Kuratov Y., Sorokin A. \(2019\). [Tuning Multilingual Transformers for Language-Specific Named Entity Recognition](https://www.aclweb.org/anthology/W19-3712/). ACL anthology W19-3712.
|
||||
@@ -1,16 +0,0 @@
|
||||
---
|
||||
language: en
|
||||
---
|
||||
|
||||
# bert-base-cased-conversational
|
||||
|
||||
Conversational BERT \(English, cased, 12‑layer, 768‑hidden, 12‑heads, 110M parameters\) was trained on the English part of Twitter, Reddit, DailyDialogues\[1\], OpenSubtitles\[2\], Debates\[3\], Blogs\[4\], Facebook News Comments. We used this training data to build the vocabulary of English subtokens and took English cased version of BERT‑base as an initialization for English Conversational BERT.
|
||||
|
||||
|
||||
\[1\]: Yanran Li, Hui Su, Xiaoyu Shen, Wenjie Li, Ziqiang Cao, and Shuzi Niu. DailyDialog: A Manually Labelled Multi-turn Dialogue Dataset. IJCNLP 2017.
|
||||
|
||||
\[2\]: P. Lison and J. Tiedemann, 2016, OpenSubtitles2016: Extracting Large Parallel Corpora from Movie and TV Subtitles. In Proceedings of the 10th International Conference on Language Resources and Evaluation \(LREC 2016\)
|
||||
|
||||
\[3\]: Justine Zhang, Ravi Kumar, Sujith Ravi, Cristian Danescu-Niculescu-Mizil. Proceedings of NAACL, 2016.
|
||||
|
||||
\[4\]: J. Schler, M. Koppel, S. Argamon and J. Pennebaker \(2006\). Effects of Age and Gender on Blogging in Proceedings of 2006 AAAI Spring Symposium on Computational Approaches for Analyzing Weblogs.
|
||||
@@ -1,15 +0,0 @@
|
||||
---
|
||||
language:
|
||||
- multilingual
|
||||
---
|
||||
|
||||
# bert-base-multilingual-cased-sentence
|
||||
|
||||
Sentence Multilingual BERT \(101 languages, cased, 12‑layer, 768‑hidden, 12‑heads, 180M parameters\) is a representation‑based sentence encoder for 101 languages of Multilingual BERT. It is initialized with Multilingual BERT and then fine‑tuned on english MultiNLI\[1\] and on dev set of multilingual XNLI\[2\]. Sentence representations are mean pooled token embeddings in the same manner as in Sentence‑BERT\[3\].
|
||||
|
||||
|
||||
\[1\]: Williams A., Nangia N. & Bowman S. \(2017\) A Broad-Coverage Challenge Corpus for Sentence Understanding through Inference. arXiv preprint [arXiv:1704.05426](https://arxiv.org/abs/1704.05426)
|
||||
|
||||
\[2\]: Williams A., Bowman S. \(2018\) XNLI: Evaluating Cross-lingual Sentence Representations. arXiv preprint [arXiv:1809.05053](https://arxiv.org/abs/1809.05053)
|
||||
|
||||
\[3\]: N. Reimers, I. Gurevych \(2019\) Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks. arXiv preprint [arXiv:1908.10084](https://arxiv.org/abs/1908.10084)
|
||||
@@ -1,13 +0,0 @@
|
||||
---
|
||||
language:
|
||||
- ru
|
||||
---
|
||||
|
||||
# rubert-base-cased-conversational
|
||||
|
||||
Conversational RuBERT \(Russian, cased, 12‑layer, 768‑hidden, 12‑heads, 180M parameters\) was trained on OpenSubtitles\[1\], [Dirty](https://d3.ru/), [Pikabu](https://pikabu.ru/), and a Social Media segment of Taiga corpus\[2\]. We assembled a new vocabulary for Conversational RuBERT model on this data and initialized the model with [RuBERT](../rubert-base-cased).
|
||||
|
||||
|
||||
\[1\]: P. Lison and J. Tiedemann, 2016, OpenSubtitles2016: Extracting Large Parallel Corpora from Movie and TV Subtitles. In Proceedings of the 10th International Conference on Language Resources and Evaluation \(LREC 2016\)
|
||||
|
||||
\[2\]: Shavrina T., Shapovalova O. \(2017\) TO THE METHODOLOGY OF CORPUS CONSTRUCTION FOR MACHINE LEARNING: «TAIGA» SYNTAX TREE CORPUS AND PARSER. in proc. of “CORPORA2017”, international conference , Saint-Petersbourg, 2017.
|
||||
@@ -1,15 +0,0 @@
|
||||
---
|
||||
language:
|
||||
- ru
|
||||
---
|
||||
|
||||
# rubert-base-cased-sentence
|
||||
|
||||
Sentence RuBERT \(Russian, cased, 12-layer, 768-hidden, 12-heads, 180M parameters\) is a representation‑based sentence encoder for Russian. It is initialized with RuBERT and fine‑tuned on SNLI\[1\] google-translated to russian and on russian part of XNLI dev set\[2\]. Sentence representations are mean pooled token embeddings in the same manner as in Sentence‑BERT\[3\].
|
||||
|
||||
|
||||
\[1\]: S. R. Bowman, G. Angeli, C. Potts, and C. D. Manning. \(2015\) A large annotated corpus for learning natural language inference. arXiv preprint [arXiv:1508.05326](https://arxiv.org/abs/1508.05326)
|
||||
|
||||
\[2\]: Williams A., Bowman S. \(2018\) XNLI: Evaluating Cross-lingual Sentence Representations. arXiv preprint [arXiv:1809.05053](https://arxiv.org/abs/1809.05053)
|
||||
|
||||
\[3\]: N. Reimers, I. Gurevych \(2019\) Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks. arXiv preprint [arXiv:1908.10084](https://arxiv.org/abs/1908.10084)
|
||||
@@ -1,11 +0,0 @@
|
||||
---
|
||||
language:
|
||||
- ru
|
||||
---
|
||||
|
||||
# rubert-base-cased
|
||||
|
||||
RuBERT \(Russian, cased, 12‑layer, 768‑hidden, 12‑heads, 180M parameters\) was trained on the Russian part of Wikipedia and news data. We used this training data to build a vocabulary of Russian subtokens and took a multilingual version of BERT‑base as an initialization for RuBERT\[1\].
|
||||
|
||||
|
||||
\[1\]: Kuratov, Y., Arkhipov, M. \(2019\). Adaptation of Deep Bidirectional Multilingual Transformers for Russian Language. arXiv preprint [arXiv:1905.07213](https://arxiv.org/abs/1905.07213).
|
||||
@@ -1,61 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
- text: "Paris est la [MASK] de la France."
|
||||
- text: "Paris est la capitale de la [MASK]."
|
||||
- text: "L'élection américaine a eu [MASK] en novembre 2020."
|
||||
- text: "تقع سويسرا في [MASK] أوروبا"
|
||||
- text: "إسمي محمد وأسكن في [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-15lang-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
The measurements below have been computed on a [Google Cloud n1-standard-1 machine (1 vCPU, 3.75 GB)](https://cloud.google.com/compute/docs/machine-types\#n1_machine_type):
|
||||
|
||||
| Model | Num parameters | Size | Memory | Loading time |
|
||||
| ------------------------------- | -------------- | -------- | -------- | ------------ |
|
||||
| bert-base-multilingual-cased | 178 million | 714 MB | 1400 MB | 4.2 sec |
|
||||
| Geotrend/bert-base-15lang-cased | 141 million | 564 MB | 1098 MB | 3.1 sec |
|
||||
|
||||
Handled languages: en, fr, es, de, zh, ar, ru, vi, el, bg, th, tr, hi, ur and sw.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-15lang-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-15lang-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: ar
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "تقع سويسرا في [MASK] أوروبا"
|
||||
- text: "إسمي محمد وأسكن في [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-ar-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-ar-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-ar-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: bg
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-bg-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-bg-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-bg-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: de
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-de-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-de-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-de-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: el
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-el-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-el-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-el-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,49 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
- text: "تقع سويسرا في [MASK] أوروبا"
|
||||
- text: "إسمي محمد وأسكن في [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-ar-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-ar-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-ar-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-bg-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-bg-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-bg-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: en
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-de-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-de-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-de-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-el-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-el-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-el-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-es-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-es-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-es-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,50 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
- text: "Paris est la [MASK] de la France."
|
||||
- text: "Paris est la capitale de la [MASK]."
|
||||
- text: "L'élection américaine a eu [MASK] en novembre 2020."
|
||||
---
|
||||
|
||||
# bert-base-en-fr-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-fr-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-fr-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-hi-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-hi-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-hi-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-ru-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-ru-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-ru-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-sw-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-sw-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-sw-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-th-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-th-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-th-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-tr-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-tr-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-tr-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-ur-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-ur-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-ur-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-vi-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-vi-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-vi-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: multilingual
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Google generated 46 billion [MASK] in revenue."
|
||||
- text: "Paris is the capital of [MASK]."
|
||||
- text: "Algiers is the largest city in [MASK]."
|
||||
---
|
||||
|
||||
# bert-base-en-zh-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-en-zh-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-en-zh-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: es
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-es-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-es-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-es-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
language: fr
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
|
||||
widget:
|
||||
- text: "Paris est la [MASK] de la France."
|
||||
- text: "Paris est la capitale de la [MASK]."
|
||||
- text: "L'élection américaine a eu [MASK] en novembre 2020."
|
||||
---
|
||||
|
||||
# bert-base-fr-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-fr-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-fr-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: hi
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-hi-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-hi-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-hi-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: ru
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-ru-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-ru-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-ru-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: sw
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-sw-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-sw-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-sw-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: th
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-th-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-th-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-th-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: tr
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-tr-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-tr-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-tr-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: ur
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-ur-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-ur-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-ur-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: vi
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-vi-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-vi-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-vi-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
language: zh
|
||||
|
||||
datasets: wikipedia
|
||||
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# bert-base-zh-cased
|
||||
|
||||
We are sharing smaller versions of [bert-base-multilingual-cased](https://huggingface.co/bert-base-multilingual-cased) that handle a custom number of languages.
|
||||
|
||||
Unlike [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased), our versions give exactly the same representations produced by the original model which preserves the original accuracy.
|
||||
|
||||
For more information please visit our paper: [Load What You Need: Smaller Versions of Multilingual BERT](https://www.aclweb.org/anthology/2020.sustainlp-1.16.pdf).
|
||||
|
||||
## How to use
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Geotrend/bert-base-zh-cased")
|
||||
model = AutoModel.from_pretrained("Geotrend/bert-base-zh-cased")
|
||||
|
||||
```
|
||||
|
||||
To generate other smaller versions of multilingual transformers please visit [our Github repo](https://github.com/Geotrend-research/smaller-transformers).
|
||||
|
||||
### How to cite
|
||||
|
||||
```bibtex
|
||||
@inproceedings{smallermbert,
|
||||
title={Load What You Need: Smaller Versions of Mutlilingual BERT},
|
||||
author={Abdaoui, Amine and Pradel, Camille and Sigel, Grégoire},
|
||||
booktitle={SustaiNLP / EMNLP},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
Please contact amine@geotrend.fr for any question, feedback or request.
|
||||
@@ -1,18 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **Arabic language**. The mono in the name refers to the monolingual setting, where the model is trained using only Arabic language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.877609 for a learning rate of 2e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **English language**. The mono in the name refers to the monolingual setting, where the model is trained using only English language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.726030 for a learning rate of 2e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **French language**. The mono in the name refers to the monolingual setting, where the model is trained using only English language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.692094 for a learning rate of 3e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **German language**. The mono in the name refers to the monolingual setting, where the model is trained using only English language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.649794 for a learning rate of 3e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **Indonesian language**. The mono in the name refers to the monolingual setting, where the model is trained using only Arabic language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.844494 for a learning rate of 2e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **Italian language**. The mono in the name refers to the monolingual setting, where the model is trained using only English language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.837288 for a learning rate of 3e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **Polish language**. The mono in the name refers to the monolingual setting, where the model is trained using only English language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.723254 for a learning rate of 2e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **Portuguese language**. The mono in the name refers to the monolingual setting, where the model is trained using only English language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.716119 for a learning rate of 3e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,20 +0,0 @@
|
||||
This model is used detecting **hatespeech** in **Spanish language**. The mono in the name refers to the monolingual setting, where the model is trained using only English language data. It is finetuned on multilingual bert model.
|
||||
The model is trained with different learning rates and the best validation score achieved is 0.740287 for a learning rate of 3e-5. Training code can be found at this [url](https://github.com/punyajoy/DE-LIMIT)
|
||||
|
||||
|
||||
|
||||
### For more details about our paper
|
||||
|
||||
Sai Saketh Aluru, Binny Mathew, Punyajoy Saha and Animesh Mukherjee. "[Deep Learning Models for Multilingual Hate Speech Detection](https://arxiv.org/abs/2004.06465)". Accepted at ECML-PKDD 2020.
|
||||
|
||||
***Please cite our paper in any published work that uses any of these resources.***
|
||||
|
||||
~~~
|
||||
@article{aluru2020deep,
|
||||
title={Deep Learning Models for Multilingual Hate Speech Detection},
|
||||
author={Aluru, Sai Saket and Mathew, Binny and Saha, Punyajoy and Mukherjee, Animesh},
|
||||
journal={arXiv preprint arXiv:2004.06465},
|
||||
year={2020}
|
||||
}
|
||||
|
||||
~~~
|
||||
@@ -1,124 +0,0 @@
|
||||
## ParsBERT: Transformer-based Model for Persian Language Understanding
|
||||
|
||||
ParsBERT is a monolingual language model based on Google’s BERT architecture with the same configurations as BERT-Base.
|
||||
|
||||
Paper presenting ParsBERT: [arXiv:2005.12515](https://arxiv.org/abs/2005.12515)
|
||||
|
||||
All the models (downstream tasks) are uncased and trained with whole word masking. (coming soon stay tuned)
|
||||
|
||||
|
||||
## Persian NER [ARMAN, PEYMA, ARMAN+PEYMA]
|
||||
|
||||
This task aims to extract named entities in the text, such as names and label with appropriate `NER` classes such as locations, organizations, etc. The datasets used for this task contain sentences that are marked with `IOB` format. In this format, tokens that are not part of an entity are tagged as `”O”` the `”B”`tag corresponds to the first word of an object, and the `”I”` tag corresponds to the rest of the terms of the same entity. Both `”B”` and `”I”` tags are followed by a hyphen (or underscore), followed by the entity category. Therefore, the NER task is a multi-class token classification problem that labels the tokens upon being fed a raw text. There are two primary datasets used in Persian NER, `ARMAN`, and `PEYMA`. In ParsBERT, we prepared ner for both datasets as well as a combination of both datasets.
|
||||
|
||||
|
||||
|
||||
### PEYMA
|
||||
|
||||
PEYMA dataset includes 7,145 sentences with a total of 302,530 tokens from which 41,148 tokens are tagged with seven different classes.
|
||||
|
||||
1. Organization
|
||||
2. Money
|
||||
3. Location
|
||||
4. Date
|
||||
5. Time
|
||||
6. Person
|
||||
7. Percent
|
||||
|
||||
|
||||
| Label | # |
|
||||
|:------------:|:-----:|
|
||||
| Organization | 16964 |
|
||||
| Money | 2037 |
|
||||
| Location | 8782 |
|
||||
| Date | 4259 |
|
||||
| Time | 732 |
|
||||
| Person | 7675 |
|
||||
| Percent | 699 |
|
||||
|
||||
|
||||
|
||||
**Download**
|
||||
You can download the dataset from [here](http://nsurl.org/tasks/task-7-named-entity-recognition-ner-for-farsi/)
|
||||
|
||||
---
|
||||
|
||||
### ARMAN
|
||||
|
||||
ARMAN dataset holds 7,682 sentences with 250,015 sentences tagged over six different classes.
|
||||
|
||||
1. Organization
|
||||
2. Location
|
||||
3. Facility
|
||||
4. Event
|
||||
5. Product
|
||||
6. Person
|
||||
|
||||
|
||||
| Label | # |
|
||||
|:------------:|:-----:|
|
||||
| Organization | 30108 |
|
||||
| Location | 12924 |
|
||||
| Facility | 4458 |
|
||||
| Event | 7557 |
|
||||
| Product | 4389 |
|
||||
| Person | 15645 |
|
||||
|
||||
|
||||
|
||||
**Download**
|
||||
You can download the dataset from [here](https://github.com/HaniehP/PersianNER)
|
||||
|
||||
|
||||
|
||||
## Results
|
||||
|
||||
The following table summarizes the F1 score obtained by ParsBERT as compared to other models and architectures.
|
||||
|
||||
| Dataset | ParsBERT | MorphoBERT | Beheshti-NER | LSTM-CRF | Rule-Based CRF | BiLSTM-CRF |
|
||||
|:---------------:|:--------:|:----------:|:--------------:|:----------:|:----------------:|:------------:|
|
||||
| ARMAN + PEYMA | 95.13* | - | - | - | - | - |
|
||||
| PEYMA | 98.79* | - | 90.59 | - | 84.00 | - |
|
||||
| ARMAN | 93.10* | 89.9 | 84.03 | 86.55 | - | 77.45 |
|
||||
|
||||
|
||||
## How to use :hugs:
|
||||
| Notebook | Description | |
|
||||
|:----------|:-------------|------:|
|
||||
| [How to use Pipelines](https://github.com/hooshvare/parsbert-ner/blob/master/persian-ner-pipeline.ipynb) | Simple and efficient way to use State-of-the-Art models on downstream tasks through transformers | [](https://colab.research.google.com/github/hooshvare/parsbert-ner/blob/master/persian-ner-pipeline.ipynb) |
|
||||
|
||||
|
||||
## Cite
|
||||
|
||||
Please cite the following paper in your publication if you are using [ParsBERT](https://arxiv.org/abs/2005.12515) in your research:
|
||||
|
||||
```markdown
|
||||
@article{ParsBERT,
|
||||
title={ParsBERT: Transformer-based Model for Persian Language Understanding},
|
||||
author={Mehrdad Farahani, Mohammad Gharachorloo, Marzieh Farahani, Mohammad Manthouri},
|
||||
journal={ArXiv},
|
||||
year={2020},
|
||||
volume={abs/2005.12515}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## Acknowledgments
|
||||
|
||||
We hereby, express our gratitude to the [Tensorflow Research Cloud (TFRC) program](https://tensorflow.org/tfrc) for providing us with the necessary computation resources. We also thank [Hooshvare](https://hooshvare.com) Research Group for facilitating dataset gathering and scraping online text resources.
|
||||
|
||||
|
||||
## Contributors
|
||||
|
||||
- Mehrdad Farahani: [Linkedin](https://www.linkedin.com/in/m3hrdadfi/), [Twitter](https://twitter.com/m3hrdadfi), [Github](https://github.com/m3hrdadfi)
|
||||
- Mohammad Gharachorloo: [Linkedin](https://www.linkedin.com/in/mohammad-gharachorloo/), [Twitter](https://twitter.com/MGharachorloo), [Github](https://github.com/baarsaam)
|
||||
- Marzieh Farahani: [Linkedin](https://www.linkedin.com/in/marziehphi/), [Twitter](https://twitter.com/marziehphi), [Github](https://github.com/marziehphi)
|
||||
- Mohammad Manthouri: [Linkedin](https://www.linkedin.com/in/mohammad-manthouri-aka-mansouri-07030766/), [Twitter](https://twitter.com/mmanthouri), [Github](https://github.com/mmanthouri)
|
||||
- Hooshvare Team: [Official Website](https://hooshvare.com/), [Linkedin](https://www.linkedin.com/company/hooshvare), [Twitter](https://twitter.com/hooshvare), [Github](https://github.com/hooshvare), [Instagram](https://www.instagram.com/hooshvare/)
|
||||
|
||||
+ And a special thanks to Sara Tabrizi for her fantastic poster design. Follow her on: [Linkedin](https://www.linkedin.com/in/sara-tabrizi-64548b79/), [Behance](https://www.behance.net/saratabrizi), [Instagram](https://www.instagram.com/sara_b_tabrizi/)
|
||||
|
||||
## Releases
|
||||
|
||||
### Release v0.1 (May 29, 2019)
|
||||
This is the first version of our ParsBERT NER!
|
||||
@@ -1,124 +0,0 @@
|
||||
## ParsBERT: Transformer-based Model for Persian Language Understanding
|
||||
|
||||
ParsBERT is a monolingual language model based on Google’s BERT architecture with the same configurations as BERT-Base.
|
||||
|
||||
Paper presenting ParsBERT: [arXiv:2005.12515](https://arxiv.org/abs/2005.12515)
|
||||
|
||||
All the models (downstream tasks) are uncased and trained with whole word masking. (coming soon stay tuned)
|
||||
|
||||
|
||||
## Persian NER [ARMAN, PEYMA, ARMAN+PEYMA]
|
||||
|
||||
This task aims to extract named entities in the text, such as names and label with appropriate `NER` classes such as locations, organizations, etc. The datasets used for this task contain sentences that are marked with `IOB` format. In this format, tokens that are not part of an entity are tagged as `”O”` the `”B”`tag corresponds to the first word of an object, and the `”I”` tag corresponds to the rest of the terms of the same entity. Both `”B”` and `”I”` tags are followed by a hyphen (or underscore), followed by the entity category. Therefore, the NER task is a multi-class token classification problem that labels the tokens upon being fed a raw text. There are two primary datasets used in Persian NER, `ARMAN`, and `PEYMA`. In ParsBERT, we prepared ner for both datasets as well as a combination of both datasets.
|
||||
|
||||
|
||||
|
||||
### PEYMA
|
||||
|
||||
PEYMA dataset includes 7,145 sentences with a total of 302,530 tokens from which 41,148 tokens are tagged with seven different classes.
|
||||
|
||||
1. Organization
|
||||
2. Money
|
||||
3. Location
|
||||
4. Date
|
||||
5. Time
|
||||
6. Person
|
||||
7. Percent
|
||||
|
||||
|
||||
| Label | # |
|
||||
|:------------:|:-----:|
|
||||
| Organization | 16964 |
|
||||
| Money | 2037 |
|
||||
| Location | 8782 |
|
||||
| Date | 4259 |
|
||||
| Time | 732 |
|
||||
| Person | 7675 |
|
||||
| Percent | 699 |
|
||||
|
||||
|
||||
|
||||
**Download**
|
||||
You can download the dataset from [here](http://nsurl.org/tasks/task-7-named-entity-recognition-ner-for-farsi/)
|
||||
|
||||
---
|
||||
|
||||
### ARMAN
|
||||
|
||||
ARMAN dataset holds 7,682 sentences with 250,015 sentences tagged over six different classes.
|
||||
|
||||
1. Organization
|
||||
2. Location
|
||||
3. Facility
|
||||
4. Event
|
||||
5. Product
|
||||
6. Person
|
||||
|
||||
|
||||
| Label | # |
|
||||
|:------------:|:-----:|
|
||||
| Organization | 30108 |
|
||||
| Location | 12924 |
|
||||
| Facility | 4458 |
|
||||
| Event | 7557 |
|
||||
| Product | 4389 |
|
||||
| Person | 15645 |
|
||||
|
||||
|
||||
|
||||
**Download**
|
||||
You can download the dataset from [here](https://github.com/HaniehP/PersianNER)
|
||||
|
||||
|
||||
|
||||
## Results
|
||||
|
||||
The following table summarizes the F1 score obtained by ParsBERT as compared to other models and architectures.
|
||||
|
||||
| Dataset | ParsBERT | MorphoBERT | Beheshti-NER | LSTM-CRF | Rule-Based CRF | BiLSTM-CRF |
|
||||
|:---------------:|:--------:|:----------:|:--------------:|:----------:|:----------------:|:------------:|
|
||||
| ARMAN + PEYMA | 95.13* | - | - | - | - | - |
|
||||
| PEYMA | 98.79* | - | 90.59 | - | 84.00 | - |
|
||||
| ARMAN | 93.10* | 89.9 | 84.03 | 86.55 | - | 77.45 |
|
||||
|
||||
|
||||
## How to use :hugs:
|
||||
| Notebook | Description | |
|
||||
|:----------|:-------------|------:|
|
||||
| [How to use Pipelines](https://github.com/hooshvare/parsbert-ner/blob/master/persian-ner-pipeline.ipynb) | Simple and efficient way to use State-of-the-Art models on downstream tasks through transformers | [](https://colab.research.google.com/github/hooshvare/parsbert-ner/blob/master/persian-ner-pipeline.ipynb) |
|
||||
|
||||
|
||||
## Cite
|
||||
|
||||
Please cite the following paper in your publication if you are using [ParsBERT](https://arxiv.org/abs/2005.12515) in your research:
|
||||
|
||||
```markdown
|
||||
@article{ParsBERT,
|
||||
title={ParsBERT: Transformer-based Model for Persian Language Understanding},
|
||||
author={Mehrdad Farahani, Mohammad Gharachorloo, Marzieh Farahani, Mohammad Manthouri},
|
||||
journal={ArXiv},
|
||||
year={2020},
|
||||
volume={abs/2005.12515}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## Acknowledgments
|
||||
|
||||
We hereby, express our gratitude to the [Tensorflow Research Cloud (TFRC) program](https://tensorflow.org/tfrc) for providing us with the necessary computation resources. We also thank [Hooshvare](https://hooshvare.com) Research Group for facilitating dataset gathering and scraping online text resources.
|
||||
|
||||
|
||||
## Contributors
|
||||
|
||||
- Mehrdad Farahani: [Linkedin](https://www.linkedin.com/in/m3hrdadfi/), [Twitter](https://twitter.com/m3hrdadfi), [Github](https://github.com/m3hrdadfi)
|
||||
- Mohammad Gharachorloo: [Linkedin](https://www.linkedin.com/in/mohammad-gharachorloo/), [Twitter](https://twitter.com/MGharachorloo), [Github](https://github.com/baarsaam)
|
||||
- Marzieh Farahani: [Linkedin](https://www.linkedin.com/in/marziehphi/), [Twitter](https://twitter.com/marziehphi), [Github](https://github.com/marziehphi)
|
||||
- Mohammad Manthouri: [Linkedin](https://www.linkedin.com/in/mohammad-manthouri-aka-mansouri-07030766/), [Twitter](https://twitter.com/mmanthouri), [Github](https://github.com/mmanthouri)
|
||||
- Hooshvare Team: [Official Website](https://hooshvare.com/), [Linkedin](https://www.linkedin.com/company/hooshvare), [Twitter](https://twitter.com/hooshvare), [Github](https://github.com/hooshvare), [Instagram](https://www.instagram.com/hooshvare/)
|
||||
|
||||
+ And a special thanks to Sara Tabrizi for her fantastic poster design. Follow her on: [Linkedin](https://www.linkedin.com/in/sara-tabrizi-64548b79/), [Behance](https://www.behance.net/saratabrizi), [Instagram](https://www.instagram.com/sara_b_tabrizi/)
|
||||
|
||||
## Releases
|
||||
|
||||
### Release v0.1 (May 29, 2019)
|
||||
This is the first version of our ParsBERT NER!
|
||||
@@ -1,124 +0,0 @@
|
||||
## ParsBERT: Transformer-based Model for Persian Language Understanding
|
||||
|
||||
ParsBERT is a monolingual language model based on Google’s BERT architecture with the same configurations as BERT-Base.
|
||||
|
||||
Paper presenting ParsBERT: [arXiv:2005.12515](https://arxiv.org/abs/2005.12515)
|
||||
|
||||
All the models (downstream tasks) are uncased and trained with whole word masking. (coming soon stay tuned)
|
||||
|
||||
|
||||
## Persian NER [ARMAN, PEYMA, ARMAN+PEYMA]
|
||||
|
||||
This task aims to extract named entities in the text, such as names and label with appropriate `NER` classes such as locations, organizations, etc. The datasets used for this task contain sentences that are marked with `IOB` format. In this format, tokens that are not part of an entity are tagged as `”O”` the `”B”`tag corresponds to the first word of an object, and the `”I”` tag corresponds to the rest of the terms of the same entity. Both `”B”` and `”I”` tags are followed by a hyphen (or underscore), followed by the entity category. Therefore, the NER task is a multi-class token classification problem that labels the tokens upon being fed a raw text. There are two primary datasets used in Persian NER, `ARMAN`, and `PEYMA`. In ParsBERT, we prepared ner for both datasets as well as a combination of both datasets.
|
||||
|
||||
|
||||
|
||||
### PEYMA
|
||||
|
||||
PEYMA dataset includes 7,145 sentences with a total of 302,530 tokens from which 41,148 tokens are tagged with seven different classes.
|
||||
|
||||
1. Organization
|
||||
2. Money
|
||||
3. Location
|
||||
4. Date
|
||||
5. Time
|
||||
6. Person
|
||||
7. Percent
|
||||
|
||||
|
||||
| Label | # |
|
||||
|:------------:|:-----:|
|
||||
| Organization | 16964 |
|
||||
| Money | 2037 |
|
||||
| Location | 8782 |
|
||||
| Date | 4259 |
|
||||
| Time | 732 |
|
||||
| Person | 7675 |
|
||||
| Percent | 699 |
|
||||
|
||||
|
||||
|
||||
**Download**
|
||||
You can download the dataset from [here](http://nsurl.org/tasks/task-7-named-entity-recognition-ner-for-farsi/)
|
||||
|
||||
---
|
||||
|
||||
### ARMAN
|
||||
|
||||
ARMAN dataset holds 7,682 sentences with 250,015 sentences tagged over six different classes.
|
||||
|
||||
1. Organization
|
||||
2. Location
|
||||
3. Facility
|
||||
4. Event
|
||||
5. Product
|
||||
6. Person
|
||||
|
||||
|
||||
| Label | # |
|
||||
|:------------:|:-----:|
|
||||
| Organization | 30108 |
|
||||
| Location | 12924 |
|
||||
| Facility | 4458 |
|
||||
| Event | 7557 |
|
||||
| Product | 4389 |
|
||||
| Person | 15645 |
|
||||
|
||||
|
||||
|
||||
**Download**
|
||||
You can download the dataset from [here](https://github.com/HaniehP/PersianNER)
|
||||
|
||||
|
||||
|
||||
## Results
|
||||
|
||||
The following table summarizes the F1 score obtained by ParsBERT as compared to other models and architectures.
|
||||
|
||||
| Dataset | ParsBERT | MorphoBERT | Beheshti-NER | LSTM-CRF | Rule-Based CRF | BiLSTM-CRF |
|
||||
|:---------------:|:--------:|:----------:|:--------------:|:----------:|:----------------:|:------------:|
|
||||
| ARMAN + PEYMA | 95.13* | - | - | - | - | - |
|
||||
| PEYMA | 98.79* | - | 90.59 | - | 84.00 | - |
|
||||
| ARMAN | 93.10* | 89.9 | 84.03 | 86.55 | - | 77.45 |
|
||||
|
||||
|
||||
## How to use :hugs:
|
||||
| Notebook | Description | |
|
||||
|:----------|:-------------|------:|
|
||||
| [How to use Pipelines](https://github.com/hooshvare/parsbert-ner/blob/master/persian-ner-pipeline.ipynb) | Simple and efficient way to use State-of-the-Art models on downstream tasks through transformers | [](https://colab.research.google.com/github/hooshvare/parsbert-ner/blob/master/persian-ner-pipeline.ipynb) |
|
||||
|
||||
|
||||
## Cite
|
||||
|
||||
Please cite the following paper in your publication if you are using [ParsBERT](https://arxiv.org/abs/2005.12515) in your research:
|
||||
|
||||
```markdown
|
||||
@article{ParsBERT,
|
||||
title={ParsBERT: Transformer-based Model for Persian Language Understanding},
|
||||
author={Mehrdad Farahani, Mohammad Gharachorloo, Marzieh Farahani, Mohammad Manthouri},
|
||||
journal={ArXiv},
|
||||
year={2020},
|
||||
volume={abs/2005.12515}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## Acknowledgments
|
||||
|
||||
We hereby, express our gratitude to the [Tensorflow Research Cloud (TFRC) program](https://tensorflow.org/tfrc) for providing us with the necessary computation resources. We also thank [Hooshvare](https://hooshvare.com) Research Group for facilitating dataset gathering and scraping online text resources.
|
||||
|
||||
|
||||
## Contributors
|
||||
|
||||
- Mehrdad Farahani: [Linkedin](https://www.linkedin.com/in/m3hrdadfi/), [Twitter](https://twitter.com/m3hrdadfi), [Github](https://github.com/m3hrdadfi)
|
||||
- Mohammad Gharachorloo: [Linkedin](https://www.linkedin.com/in/mohammad-gharachorloo/), [Twitter](https://twitter.com/MGharachorloo), [Github](https://github.com/baarsaam)
|
||||
- Marzieh Farahani: [Linkedin](https://www.linkedin.com/in/marziehphi/), [Twitter](https://twitter.com/marziehphi), [Github](https://github.com/marziehphi)
|
||||
- Mohammad Manthouri: [Linkedin](https://www.linkedin.com/in/mohammad-manthouri-aka-mansouri-07030766/), [Twitter](https://twitter.com/mmanthouri), [Github](https://github.com/mmanthouri)
|
||||
- Hooshvare Team: [Official Website](https://hooshvare.com/), [Linkedin](https://www.linkedin.com/company/hooshvare), [Twitter](https://twitter.com/hooshvare), [Github](https://github.com/hooshvare), [Instagram](https://www.instagram.com/hooshvare/)
|
||||
|
||||
+ And a special thanks to Sara Tabrizi for her fantastic poster design. Follow her on: [Linkedin](https://www.linkedin.com/in/sara-tabrizi-64548b79/), [Behance](https://www.behance.net/saratabrizi), [Instagram](https://www.instagram.com/sara_b_tabrizi/)
|
||||
|
||||
## Releases
|
||||
|
||||
### Release v0.1 (May 29, 2019)
|
||||
This is the first version of our ParsBERT NER!
|
||||
@@ -1,124 +0,0 @@
|
||||
## ParsBERT: Transformer-based Model for Persian Language Understanding
|
||||
|
||||
ParsBERT is a monolingual language model based on Google’s BERT architecture with the same configurations as BERT-Base.
|
||||
|
||||
Paper presenting ParsBERT: [arXiv:2005.12515](https://arxiv.org/abs/2005.12515)
|
||||
|
||||
All the models (downstream tasks) are uncased and trained with whole word masking. (coming soon stay tuned)
|
||||
|
||||
|
||||
---
|
||||
|
||||
## Introduction
|
||||
|
||||
This model is pre-trained on a large Persian corpus with various writing styles from numerous subjects (e.g., scientific, novels, news) with more than 2M documents. A large subset of this corpus was crawled manually.
|
||||
|
||||
As a part of ParsBERT methodology, an extensive pre-processing combining POS tagging and WordPiece segmentation was carried out to bring the corpus into a proper format. This process produces more than 40M true sentences.
|
||||
|
||||
|
||||
## Evaluation
|
||||
|
||||
ParsBERT is evaluated on three NLP downstream tasks: Sentiment Analysis (SA), Text Classification, and Named Entity Recognition (NER). For this matter and due to insufficient resources, two large datasets for SA and two for text classification were manually composed, which are available for public use and benchmarking. ParsBERT outperformed all other language models, including multilingual BERT and other hybrid deep learning models for all tasks, improving the state-of-the-art performance in Persian language modeling.
|
||||
|
||||
## Results
|
||||
|
||||
The following table summarizes the F1 score obtained by ParsBERT as compared to other models and architectures.
|
||||
|
||||
|
||||
|
||||
### Sentiment Analysis (SA) task
|
||||
|
||||
| Dataset | ParsBERT | mBERT | DeepSentiPers |
|
||||
|:--------------------------:|:---------:|:-----:|:-------------:|
|
||||
| Digikala User Comments | 81.74* | 80.74 | - |
|
||||
| SnappFood User Comments | 88.12* | 87.87 | - |
|
||||
| SentiPers (Multi Class) | 71.11* | - | 69.33 |
|
||||
| SentiPers (Binary Class) | 92.13* | - | 91.98 |
|
||||
|
||||
|
||||
|
||||
### Text Classification (TC) task
|
||||
|
||||
| Dataset | ParsBERT | mBERT |
|
||||
|:-----------------:|:--------:|:-----:|
|
||||
| Digikala Magazine | 93.59* | 90.72 |
|
||||
| Persian News | 97.19* | 95.79 |
|
||||
|
||||
|
||||
### Named Entity Recognition (NER) task
|
||||
|
||||
| Dataset | ParsBERT | mBERT | MorphoBERT | Beheshti-NER | LSTM-CRF | Rule-Based CRF | BiLSTM-CRF |
|
||||
|:-------:|:--------:|:--------:|:----------:|:--------------:|:----------:|:----------------:|:------------:|
|
||||
| PEYMA | 93.10* | 86.64 | - | 90.59 | - | 84.00 | - |
|
||||
| ARMAN | 98.79* | 95.89 | 89.9 | 84.03 | 86.55 | - | 77.45 |
|
||||
|
||||
|
||||
**If you tested ParsBERT on a public dataset and you want to add your results to the table above, open a pull request or contact us. Also make sure to have your code available online so we can add it as a reference**
|
||||
|
||||
## How to use
|
||||
|
||||
### TensorFlow 2.0
|
||||
|
||||
```python
|
||||
from transformers import AutoConfig, AutoTokenizer, TFAutoModel
|
||||
|
||||
config = AutoConfig.from_pretrained("HooshvareLab/bert-base-parsbert-uncased")
|
||||
tokenizer = AutoTokenizer.from_pretrained("HooshvareLab/bert-base-parsbert-uncased")
|
||||
model = AutoModel.from_pretrained("HooshvareLab/bert-base-parsbert-uncased")
|
||||
|
||||
text = "ما در هوشواره معتقدیم با انتقال صحیح دانش و آگاهی، همه افراد میتوانند از ابزارهای هوشمند استفاده کنند. شعار ما هوش مصنوعی برای همه است."
|
||||
tokenizer.tokenize(text)
|
||||
|
||||
>>> ['ما', 'در', 'هوش', '##واره', 'معتقدیم', 'با', 'انتقال', 'صحیح', 'دانش', 'و', 'اگاهی', '،', 'همه', 'افراد', 'میتوانند', 'از', 'ابزارهای', 'هوشمند', 'استفاده', 'کنند', '.', 'شعار', 'ما', 'هوش', 'مصنوعی', 'برای', 'همه', 'است', '.']
|
||||
|
||||
```
|
||||
|
||||
### Pytorch
|
||||
|
||||
```python
|
||||
from transformers import AutoConfig, AutoTokenizer, AutoModel
|
||||
|
||||
config = AutoConfig.from_pretrained("HooshvareLab/bert-base-parsbert-uncased")
|
||||
tokenizer = AutoTokenizer.from_pretrained("HooshvareLab/bert-base-parsbert-uncased")
|
||||
model = AutoModel.from_pretrained("HooshvareLab/bert-base-parsbert-uncased")
|
||||
```
|
||||
|
||||
|
||||
## NLP Tasks Tutorial
|
||||
|
||||
Coming soon stay tuned
|
||||
|
||||
|
||||
## Cite
|
||||
|
||||
Please cite the following paper in your publication if you are using [ParsBERT](https://arxiv.org/abs/2005.12515) in your research:
|
||||
|
||||
```markdown
|
||||
@article{ParsBERT,
|
||||
title={ParsBERT: Transformer-based Model for Persian Language Understanding},
|
||||
author={Mehrdad Farahani, Mohammad Gharachorloo, Marzieh Farahani, Mohammad Manthouri},
|
||||
journal={ArXiv},
|
||||
year={2020},
|
||||
volume={abs/2005.12515}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## Acknowledgments
|
||||
|
||||
We hereby, express our gratitude to the [Tensorflow Research Cloud (TFRC) program](https://tensorflow.org/tfrc) for providing us with the necessary computation resources. We also thank [Hooshvare](https://hooshvare.com) Research Group for facilitating dataset gathering and scraping online text resources.
|
||||
|
||||
|
||||
## Contributors
|
||||
|
||||
- Mehrdad Farahani: [Linkedin](https://www.linkedin.com/in/m3hrdadfi/), [Twitter](https://twitter.com/m3hrdadfi), [Github](https://github.com/m3hrdadfi)
|
||||
- Mohammad Gharachorloo: [Linkedin](https://www.linkedin.com/in/mohammad-gharachorloo/), [Twitter](https://twitter.com/MGharachorloo), [Github](https://github.com/baarsaam)
|
||||
- Marzieh Farahani: [Linkedin](https://www.linkedin.com/in/marziehphi/), [Twitter](https://twitter.com/marziehphi), [Github](https://github.com/marziehphi)
|
||||
- Mohammad Manthouri: [Linkedin](https://www.linkedin.com/in/mohammad-manthouri-aka-mansouri-07030766/), [Twitter](https://twitter.com/mmanthouri), [Github](https://github.com/mmanthouri)
|
||||
- Hooshvare Team: [Official Website](https://hooshvare.com/), [Linkedin](https://www.linkedin.com/company/hooshvare), [Twitter](https://twitter.com/hooshvare), [Github](https://github.com/hooshvare), [Instagram](https://www.instagram.com/hooshvare/)
|
||||
|
||||
|
||||
## Releases
|
||||
|
||||
### Release v0.1 (May 27, 2019)
|
||||
This is the first version of our ParsBERT based on BERT<sub>BASE</sub>
|
||||
@@ -1,147 +0,0 @@
|
||||
---
|
||||
language: fa
|
||||
tags:
|
||||
- bert-fa
|
||||
- bert-persian
|
||||
- persian-lm
|
||||
license: apache-2.0
|
||||
---
|
||||
|
||||
# ParsBERT (v2.0)
|
||||
A Transformer-based Model for Persian Language Understanding
|
||||
|
||||
|
||||
We reconstructed the vocabulary and fine-tuned the ParsBERT v1.1 on the new Persian corpora in order to provide some functionalities for using ParsBERT in other scopes!
|
||||
Please follow the [ParsBERT](https://github.com/hooshvare/parsbert) repo for the latest information about previous and current models.
|
||||
|
||||
## Introduction
|
||||
|
||||
ParsBERT is a monolingual language model based on Google’s BERT architecture. This model is pre-trained on large Persian corpora with various writing styles from numerous subjects (e.g., scientific, novels, news) with more than `3.9M` documents, `73M` sentences, and `1.3B` words.
|
||||
|
||||
Paper presenting ParsBERT: [arXiv:2005.12515](https://arxiv.org/abs/2005.12515)
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
You can use the raw model for either masked language modeling or next sentence prediction, but it's mostly intended to
|
||||
be fine-tuned on a downstream task. See the [model hub](https://huggingface.co/models?search=bert-fa) to look for
|
||||
fine-tuned versions on a task that interests you.
|
||||
|
||||
|
||||
### How to use
|
||||
|
||||
#### TensorFlow 2.0
|
||||
|
||||
```python
|
||||
from transformers import AutoConfig, AutoTokenizer, TFAutoModel
|
||||
|
||||
config = AutoConfig.from_pretrained("HooshvareLab/bert-fa-base-uncased")
|
||||
tokenizer = AutoTokenizer.from_pretrained("HooshvareLab/bert-fa-base-uncased")
|
||||
model = TFAutoModel.from_pretrained("HooshvareLab/bert-fa-base-uncased")
|
||||
|
||||
text = "ما در هوشواره معتقدیم با انتقال صحیح دانش و آگاهی، همه افراد میتوانند از ابزارهای هوشمند استفاده کنند. شعار ما هوش مصنوعی برای همه است."
|
||||
tokenizer.tokenize(text)
|
||||
|
||||
>>> ['ما', 'در', 'هوش', '##واره', 'معتقدیم', 'با', 'انتقال', 'صحیح', 'دانش', 'و', 'اگاهی', '،', 'همه', 'افراد', 'میتوانند', 'از', 'ابزارهای', 'هوشمند', 'استفاده', 'کنند', '.', 'شعار', 'ما', 'هوش', 'مصنوعی', 'برای', 'همه', 'است', '.']
|
||||
```
|
||||
|
||||
#### Pytorch
|
||||
|
||||
```python
|
||||
from transformers import AutoConfig, AutoTokenizer, AutoModel
|
||||
|
||||
config = AutoConfig.from_pretrained("HooshvareLab/bert-fa-base-uncased")
|
||||
tokenizer = AutoTokenizer.from_pretrained("HooshvareLab/bert-fa-base-uncased")
|
||||
model = AutoModel.from_pretrained("HooshvareLab/bert-fa-base-uncased")
|
||||
```
|
||||
|
||||
## Training
|
||||
|
||||
ParsBERT trained on a massive amount of public corpora ([Persian Wikidumps](https://dumps.wikimedia.org/fawiki/), [MirasText](https://github.com/miras-tech/MirasText)) and six other manually crawled text data from a various type of websites ([BigBang Page](https://bigbangpage.com/) `scientific`, [Chetor](https://www.chetor.com/) `lifestyle`, [Eligasht](https://www.eligasht.com/Blog/) `itinerary`, [Digikala](https://www.digikala.com/mag/) `digital magazine`, [Ted Talks](https://www.ted.com/talks) `general conversational`, Books `novels, storybooks, short stories from old to the contemporary era`).
|
||||
|
||||
As a part of ParsBERT methodology, an extensive pre-processing combining POS tagging and WordPiece segmentation was carried out to bring the corpora into a proper format.
|
||||
|
||||
## Goals
|
||||
Objective goals during training are as below (after 300k steps).
|
||||
|
||||
``` bash
|
||||
***** Eval results *****
|
||||
global_step = 300000
|
||||
loss = 1.4392426
|
||||
masked_lm_accuracy = 0.6865794
|
||||
masked_lm_loss = 1.4469004
|
||||
next_sentence_accuracy = 1.0
|
||||
next_sentence_loss = 6.534152e-05
|
||||
```
|
||||
|
||||
|
||||
## Derivative models
|
||||
|
||||
### Base Config
|
||||
|
||||
#### ParsBERT v2.0 Model
|
||||
- [HooshvareLab/bert-fa-base-uncased](https://huggingface.co/HooshvareLab/bert-fa-base-uncased)
|
||||
|
||||
#### ParsBERT v2.0 Sentiment Analysis
|
||||
- [HooshvareLab/bert-fa-base-uncased-sentiment-digikala](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-sentiment-digikala)
|
||||
- [HooshvareLab/bert-fa-base-uncased-sentiment-snappfood](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-sentiment-snappfood)
|
||||
- [HooshvareLab/bert-fa-base-uncased-sentiment-deepsentipers-binary](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-sentiment-deepsentipers-binary)
|
||||
- [HooshvareLab/bert-fa-base-uncased-sentiment-deepsentipers-multi](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-sentiment-deepsentipers-multi)
|
||||
|
||||
#### ParsBERT v2.0 Text Classification
|
||||
- [HooshvareLab/bert-fa-base-uncased-clf-digimag](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-clf-digimag)
|
||||
- [HooshvareLab/bert-fa-base-uncased-clf-persiannews](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-clf-persiannews)
|
||||
|
||||
#### ParsBERT v2.0 NER
|
||||
- [HooshvareLab/bert-fa-base-uncased-ner-peyma](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-ner-peyma)
|
||||
- [HooshvareLab/bert-fa-base-uncased-ner-arman](https://huggingface.co/HooshvareLab/bert-fa-base-uncased-ner-arman)
|
||||
|
||||
|
||||
## Eval results
|
||||
|
||||
ParsBERT is evaluated on three NLP downstream tasks: Sentiment Analysis (SA), Text Classification, and Named Entity Recognition (NER). For this matter and due to insufficient resources, two large datasets for SA and two for text classification were manually composed, which are available for public use and benchmarking. ParsBERT outperformed all other language models, including multilingual BERT and other hybrid deep learning models for all tasks, improving the state-of-the-art performance in Persian language modeling.
|
||||
|
||||
|
||||
### Sentiment Analysis (SA) Task
|
||||
|
||||
| Dataset | ParsBERT v2 | ParsBERT v1 | mBERT | DeepSentiPers |
|
||||
|:------------------------:|:-----------:|:-----------:|:-----:|:-------------:|
|
||||
| Digikala User Comments | 81.72 | 81.74* | 80.74 | - |
|
||||
| SnappFood User Comments | 87.98 | 88.12* | 87.87 | - |
|
||||
| SentiPers (Multi Class) | 71.31* | 71.11 | - | 69.33 |
|
||||
| SentiPers (Binary Class) | 92.42* | 92.13 | - | 91.98 |
|
||||
|
||||
|
||||
### Text Classification (TC) Task
|
||||
|
||||
| Dataset | ParsBERT v2 | ParsBERT v1 | mBERT |
|
||||
|:-----------------:|:-----------:|:-----------:|:-----:|
|
||||
| Digikala Magazine | 93.65* | 93.59 | 90.72 |
|
||||
| Persian News | 97.44* | 97.19 | 95.79 |
|
||||
|
||||
|
||||
### Named Entity Recognition (NER) Task
|
||||
|
||||
| Dataset | ParsBERT v2 | ParsBERT v1 | mBERT | MorphoBERT | Beheshti-NER | LSTM-CRF | Rule-Based CRF | BiLSTM-CRF |
|
||||
|:-------:|:-----------:|:-----------:|:-----:|:----------:|:------------:|:--------:|:--------------:|:----------:|
|
||||
| PEYMA | 93.40* | 93.10 | 86.64 | - | 90.59 | - | 84.00 | - |
|
||||
| ARMAN | 99.84* | 98.79 | 95.89 | 89.9 | 84.03 | 86.55 | - | 77.45 |
|
||||
|
||||
|
||||
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
Please cite in publications as the following:
|
||||
|
||||
```bibtex
|
||||
@article{ParsBERT,
|
||||
title={ParsBERT: Transformer-based Model for Persian Language Understanding},
|
||||
author={Mehrdad Farahani, Mohammad Gharachorloo, Marzieh Farahani, Mohammad Manthouri},
|
||||
journal={ArXiv},
|
||||
year={2020},
|
||||
volume={abs/2005.12515}
|
||||
}
|
||||
```
|
||||
|
||||
## Questions?
|
||||
Post a Github issue on the [ParsBERT Issues](https://github.com/hooshvare/parsbert/issues) repo.
|
||||
@@ -1,121 +0,0 @@
|
||||
---
|
||||
language: sv
|
||||
---
|
||||
|
||||
# Swedish BERT Models
|
||||
|
||||
The National Library of Sweden / KBLab releases three pretrained language models based on BERT and ALBERT. The models are trained on approximately 15-20GB of text (200M sentences, 3000M tokens) from various sources (books, news, government publications, swedish wikipedia and internet forums) aiming to provide a representative BERT model for Swedish text. A more complete description will be published later on.
|
||||
|
||||
The following three models are currently available:
|
||||
|
||||
- **bert-base-swedish-cased** (*v1*) - A BERT trained with the same hyperparameters as first published by Google.
|
||||
- **bert-base-swedish-cased-ner** (*experimental*) - a BERT fine-tuned for NER using SUC 3.0.
|
||||
- **albert-base-swedish-cased-alpha** (*alpha*) - A first attempt at an ALBERT for Swedish.
|
||||
|
||||
All models are cased and trained with whole word masking.
|
||||
|
||||
## Files
|
||||
|
||||
| **name** | **files** |
|
||||
|---------------------------------|-----------|
|
||||
| bert-base-swedish-cased | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/config.json), [vocab](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/vocab.txt), [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/pytorch_model.bin) |
|
||||
| bert-base-swedish-cased-ner | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/config.json), [vocab](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/vocab.txt) [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/pytorch_model.bin) |
|
||||
| albert-base-swedish-cased-alpha | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/config.json), [sentencepiece model](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/spiece.model), [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/pytorch_model.bin) |
|
||||
|
||||
TensorFlow model weights will be released soon.
|
||||
|
||||
## Usage requirements / installation instructions
|
||||
|
||||
The examples below require Huggingface Transformers 2.4.1 and Pytorch 1.3.1 or greater. For Transformers<2.4.0 the tokenizer must be instantiated manually and the `do_lower_case` flag parameter set to `False` and `keep_accents` to `True` (for ALBERT).
|
||||
|
||||
To create an environment where the examples can be run, run the following in an terminal on your OS of choice.
|
||||
|
||||
```
|
||||
# git clone https://github.com/Kungbib/swedish-bert-models
|
||||
# cd swedish-bert-models
|
||||
# python3 -m venv venv
|
||||
# source venv/bin/activate
|
||||
# pip install --upgrade pip
|
||||
# pip install -r requirements.txt
|
||||
```
|
||||
|
||||
### BERT Base Swedish
|
||||
|
||||
A standard BERT base for Swedish trained on a variety of sources. Vocabulary size is ~50k. Using Huggingface Transformers the model can be loaded in Python as follows:
|
||||
|
||||
```python
|
||||
from transformers import AutoModel,AutoTokenizer
|
||||
|
||||
tok = AutoTokenizer.from_pretrained('KB/bert-base-swedish-cased')
|
||||
model = AutoModel.from_pretrained('KB/bert-base-swedish-cased')
|
||||
```
|
||||
|
||||
|
||||
### BERT base fine-tuned for Swedish NER
|
||||
|
||||
This model is fine-tuned on the SUC 3.0 dataset. Using the Huggingface pipeline the model can be easily instantiated. For Transformer<2.4.1 it seems the tokenizer must be loaded separately to disable lower-casing of input strings:
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
nlp = pipeline('ner', model='KB/bert-base-swedish-cased-ner', tokenizer='KB/bert-base-swedish-cased-ner')
|
||||
|
||||
nlp('Idag släpper KB tre språkmodeller.')
|
||||
```
|
||||
|
||||
Running the Python code above should produce in something like the result below. Entity types used are `TME` for time, `PRS` for personal names, `LOC` for locations, `EVN` for events and `ORG` for organisations. These labels are subject to change.
|
||||
|
||||
```python
|
||||
[ { 'word': 'Idag', 'score': 0.9998126029968262, 'entity': 'TME' },
|
||||
{ 'word': 'KB', 'score': 0.9814832210540771, 'entity': 'ORG' } ]
|
||||
```
|
||||
|
||||
The BERT tokenizer often splits words into multiple tokens, with the subparts starting with `##`, for example the string `Engelbert kör Volvo till Herrängens fotbollsklubb` gets tokenized as `Engel ##bert kör Volvo till Herr ##ängens fotbolls ##klubb`. To glue parts back together one can use something like this:
|
||||
|
||||
```python
|
||||
text = 'Engelbert tar Volvon till Tele2 Arena för att titta på Djurgården IF ' +\
|
||||
'som spelar fotboll i VM klockan två på kvällen.'
|
||||
|
||||
l = []
|
||||
for token in nlp(text):
|
||||
if token['word'].startswith('##'):
|
||||
l[-1]['word'] += token['word'][2:]
|
||||
else:
|
||||
l += [ token ]
|
||||
|
||||
print(l)
|
||||
```
|
||||
|
||||
Which should result in the following (though less cleanly formatted):
|
||||
|
||||
```python
|
||||
[ { 'word': 'Engelbert', 'score': 0.99..., 'entity': 'PRS'},
|
||||
{ 'word': 'Volvon', 'score': 0.99..., 'entity': 'OBJ'},
|
||||
{ 'word': 'Tele2', 'score': 0.99..., 'entity': 'LOC'},
|
||||
{ 'word': 'Arena', 'score': 0.99..., 'entity': 'LOC'},
|
||||
{ 'word': 'Djurgården', 'score': 0.99..., 'entity': 'ORG'},
|
||||
{ 'word': 'IF', 'score': 0.99..., 'entity': 'ORG'},
|
||||
{ 'word': 'VM', 'score': 0.99..., 'entity': 'EVN'},
|
||||
{ 'word': 'klockan', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'två', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'på', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'kvällen', 'score': 0.54..., 'entity': 'TME'} ]
|
||||
```
|
||||
|
||||
### ALBERT base
|
||||
|
||||
The easiest way to do this is, again, using Huggingface Transformers:
|
||||
|
||||
```python
|
||||
from transformers import AutoModel,AutoTokenizer
|
||||
|
||||
tok = AutoTokenizer.from_pretrained('KB/albert-base-swedish-cased-alpha'),
|
||||
model = AutoModel.from_pretrained('KB/albert-base-swedish-cased-alpha')
|
||||
```
|
||||
|
||||
## Acknowledgements ❤️
|
||||
|
||||
- Resources from Stockholms University, Umeå University and Swedish Language Bank at Gothenburg University were used when fine-tuning BERT for NER.
|
||||
- Model pretraining was made partly in-house at the KBLab and partly (for material without active copyright) with the support of Cloud TPUs from Google's TensorFlow Research Cloud (TFRC).
|
||||
- Models are hosted on S3 by Huggingface 🤗
|
||||
|
||||
@@ -1,121 +0,0 @@
|
||||
---
|
||||
language: sv
|
||||
---
|
||||
|
||||
# Swedish BERT Models
|
||||
|
||||
The National Library of Sweden / KBLab releases three pretrained language models based on BERT and ALBERT. The models are trained on approximately 15-20GB of text (200M sentences, 3000M tokens) from various sources (books, news, government publications, swedish wikipedia and internet forums) aiming to provide a representative BERT model for Swedish text. A more complete description will be published later on.
|
||||
|
||||
The following three models are currently available:
|
||||
|
||||
- **bert-base-swedish-cased** (*v1*) - A BERT trained with the same hyperparameters as first published by Google.
|
||||
- **bert-base-swedish-cased-ner** (*experimental*) - a BERT fine-tuned for NER using SUC 3.0.
|
||||
- **albert-base-swedish-cased-alpha** (*alpha*) - A first attempt at an ALBERT for Swedish.
|
||||
|
||||
All models are cased and trained with whole word masking.
|
||||
|
||||
## Files
|
||||
|
||||
| **name** | **files** |
|
||||
|---------------------------------|-----------|
|
||||
| bert-base-swedish-cased | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/config.json), [vocab](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/vocab.txt), [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/pytorch_model.bin) |
|
||||
| bert-base-swedish-cased-ner | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/config.json), [vocab](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/vocab.txt) [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/pytorch_model.bin) |
|
||||
| albert-base-swedish-cased-alpha | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/config.json), [sentencepiece model](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/spiece.model), [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/pytorch_model.bin) |
|
||||
|
||||
TensorFlow model weights will be released soon.
|
||||
|
||||
## Usage requirements / installation instructions
|
||||
|
||||
The examples below require Huggingface Transformers 2.4.1 and Pytorch 1.3.1 or greater. For Transformers<2.4.0 the tokenizer must be instantiated manually and the `do_lower_case` flag parameter set to `False` and `keep_accents` to `True` (for ALBERT).
|
||||
|
||||
To create an environment where the examples can be run, run the following in an terminal on your OS of choice.
|
||||
|
||||
```
|
||||
# git clone https://github.com/Kungbib/swedish-bert-models
|
||||
# cd swedish-bert-models
|
||||
# python3 -m venv venv
|
||||
# source venv/bin/activate
|
||||
# pip install --upgrade pip
|
||||
# pip install -r requirements.txt
|
||||
```
|
||||
|
||||
### BERT Base Swedish
|
||||
|
||||
A standard BERT base for Swedish trained on a variety of sources. Vocabulary size is ~50k. Using Huggingface Transformers the model can be loaded in Python as follows:
|
||||
|
||||
```python
|
||||
from transformers import AutoModel,AutoTokenizer
|
||||
|
||||
tok = AutoTokenizer.from_pretrained('KB/bert-base-swedish-cased')
|
||||
model = AutoModel.from_pretrained('KB/bert-base-swedish-cased')
|
||||
```
|
||||
|
||||
|
||||
### BERT base fine-tuned for Swedish NER
|
||||
|
||||
This model is fine-tuned on the SUC 3.0 dataset. Using the Huggingface pipeline the model can be easily instantiated. For Transformer<2.4.1 it seems the tokenizer must be loaded separately to disable lower-casing of input strings:
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
nlp = pipeline('ner', model='KB/bert-base-swedish-cased-ner', tokenizer='KB/bert-base-swedish-cased-ner')
|
||||
|
||||
nlp('Idag släpper KB tre språkmodeller.')
|
||||
```
|
||||
|
||||
Running the Python code above should produce in something like the result below. Entity types used are `TME` for time, `PRS` for personal names, `LOC` for locations, `EVN` for events and `ORG` for organisations. These labels are subject to change.
|
||||
|
||||
```python
|
||||
[ { 'word': 'Idag', 'score': 0.9998126029968262, 'entity': 'TME' },
|
||||
{ 'word': 'KB', 'score': 0.9814832210540771, 'entity': 'ORG' } ]
|
||||
```
|
||||
|
||||
The BERT tokenizer often splits words into multiple tokens, with the subparts starting with `##`, for example the string `Engelbert kör Volvo till Herrängens fotbollsklubb` gets tokenized as `Engel ##bert kör Volvo till Herr ##ängens fotbolls ##klubb`. To glue parts back together one can use something like this:
|
||||
|
||||
```python
|
||||
text = 'Engelbert tar Volvon till Tele2 Arena för att titta på Djurgården IF ' +\
|
||||
'som spelar fotboll i VM klockan två på kvällen.'
|
||||
|
||||
l = []
|
||||
for token in nlp(text):
|
||||
if token['word'].startswith('##'):
|
||||
l[-1]['word'] += token['word'][2:]
|
||||
else:
|
||||
l += [ token ]
|
||||
|
||||
print(l)
|
||||
```
|
||||
|
||||
Which should result in the following (though less cleanly formatted):
|
||||
|
||||
```python
|
||||
[ { 'word': 'Engelbert', 'score': 0.99..., 'entity': 'PRS'},
|
||||
{ 'word': 'Volvon', 'score': 0.99..., 'entity': 'OBJ'},
|
||||
{ 'word': 'Tele2', 'score': 0.99..., 'entity': 'LOC'},
|
||||
{ 'word': 'Arena', 'score': 0.99..., 'entity': 'LOC'},
|
||||
{ 'word': 'Djurgården', 'score': 0.99..., 'entity': 'ORG'},
|
||||
{ 'word': 'IF', 'score': 0.99..., 'entity': 'ORG'},
|
||||
{ 'word': 'VM', 'score': 0.99..., 'entity': 'EVN'},
|
||||
{ 'word': 'klockan', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'två', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'på', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'kvällen', 'score': 0.54..., 'entity': 'TME'} ]
|
||||
```
|
||||
|
||||
### ALBERT base
|
||||
|
||||
The easiest way to do this is, again, using Huggingface Transformers:
|
||||
|
||||
```python
|
||||
from transformers import AutoModel,AutoTokenizer
|
||||
|
||||
tok = AutoTokenizer.from_pretrained('KB/albert-base-swedish-cased-alpha'),
|
||||
model = AutoModel.from_pretrained('KB/albert-base-swedish-cased-alpha')
|
||||
```
|
||||
|
||||
## Acknowledgements ❤️
|
||||
|
||||
- Resources from Stockholms University, Umeå University and Swedish Language Bank at Gothenburg University were used when fine-tuning BERT for NER.
|
||||
- Model pretraining was made partly in-house at the KBLab and partly (for material without active copyright) with the support of Cloud TPUs from Google's TensorFlow Research Cloud (TFRC).
|
||||
- Models are hosted on S3 by Huggingface 🤗
|
||||
|
||||
@@ -1,121 +0,0 @@
|
||||
---
|
||||
language: sv
|
||||
---
|
||||
|
||||
# Swedish BERT Models
|
||||
|
||||
The National Library of Sweden / KBLab releases three pretrained language models based on BERT and ALBERT. The models are trained on aproximately 15-20GB of text (200M sentences, 3000M tokens) from various sources (books, news, government publications, swedish wikipedia and internet forums) aiming to provide a representative BERT model for Swedish text. A more complete description will be published later on.
|
||||
|
||||
The following three models are currently available:
|
||||
|
||||
- **bert-base-swedish-cased** (*v1*) - A BERT trained with the same hyperparameters as first published by Google.
|
||||
- **bert-base-swedish-cased-ner** (*experimental*) - a BERT fine-tuned for NER using SUC 3.0.
|
||||
- **albert-base-swedish-cased-alpha** (*alpha*) - A first attempt at an ALBERT for Swedish.
|
||||
|
||||
All models are cased and trained with whole word masking.
|
||||
|
||||
## Files
|
||||
|
||||
| **name** | **files** |
|
||||
|---------------------------------|-----------|
|
||||
| bert-base-swedish-cased | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/config.json), [vocab](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/vocab.txt), [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased/pytorch_model.bin) |
|
||||
| bert-base-swedish-cased-ner | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/config.json), [vocab](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/vocab.txt) [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/bert-base-swedish-cased-ner/pytorch_model.bin) |
|
||||
| albert-base-swedish-cased-alpha | [config](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/config.json), [sentencepiece model](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/spiece.model), [pytorch_model.bin](https://s3.amazonaws.com/models.huggingface.co/bert/KB/albert-base-swedish-cased-alpha/pytorch_model.bin) |
|
||||
|
||||
TensorFlow model weights will be released soon.
|
||||
|
||||
## Usage requirements / installation instructions
|
||||
|
||||
The examples below require Huggingface Transformers 2.4.1 and Pytorch 1.3.1 or greater. For Transformers<2.4.0 the tokenizer must be instantiated manually and the `do_lower_case` flag parameter set to `False` and `keep_accents` to `True` (for ALBERT).
|
||||
|
||||
To create an environment where the examples can be run, run the following in an terminal on your OS of choice.
|
||||
|
||||
```
|
||||
# git clone https://github.com/Kungbib/swedish-bert-models
|
||||
# cd swedish-bert-models
|
||||
# python3 -m venv venv
|
||||
# source venv/bin/activate
|
||||
# pip install --upgrade pip
|
||||
# pip install -r requirements.txt
|
||||
```
|
||||
|
||||
### BERT Base Swedish
|
||||
|
||||
A standard BERT base for Swedish trained on a variety of sources. Vocabulary size is ~50k. Using Huggingface Transformers the model can be loaded in Python as follows:
|
||||
|
||||
```python
|
||||
from transformers import AutoModel,AutoTokenizer
|
||||
|
||||
tok = AutoTokenizer.from_pretrained('KB/bert-base-swedish-cased')
|
||||
model = AutoModel.from_pretrained('KB/bert-base-swedish-cased')
|
||||
```
|
||||
|
||||
|
||||
### BERT base fine-tuned for Swedish NER
|
||||
|
||||
This model is fine-tuned on the SUC 3.0 dataset. Using the Huggingface pipeline the model can be easily instantiated. For Transformer<2.4.1 it seems the tokenizer must be loaded separately to disable lower-casing of input strings:
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
nlp = pipeline('ner', model='KB/bert-base-swedish-cased-ner', tokenizer='KB/bert-base-swedish-cased-ner')
|
||||
|
||||
nlp('Idag släpper KB tre språkmodeller.')
|
||||
```
|
||||
|
||||
Running the Python code above should produce in something like the result below. Entity types used are `TME` for time, `PRS` for personal names, `LOC` for locations, `EVN` for events and `ORG` for organisations. These labels are subject to change.
|
||||
|
||||
```python
|
||||
[ { 'word': 'Idag', 'score': 0.9998126029968262, 'entity': 'TME' },
|
||||
{ 'word': 'KB', 'score': 0.9814832210540771, 'entity': 'ORG' } ]
|
||||
```
|
||||
|
||||
The BERT tokenizer often splits words into multiple tokens, with the subparts starting with `##`, for example the string `Engelbert kör Volvo till Herrängens fotbollsklubb` gets tokenized as `Engel ##bert kör Volvo till Herr ##ängens fotbolls ##klubb`. To glue parts back together one can use something like this:
|
||||
|
||||
```python
|
||||
text = 'Engelbert tar Volvon till Tele2 Arena för att titta på Djurgården IF ' +\
|
||||
'som spelar fotboll i VM klockan två på kvällen.'
|
||||
|
||||
l = []
|
||||
for token in nlp(text):
|
||||
if token['word'].startswith('##'):
|
||||
l[-1]['word'] += token['word'][2:]
|
||||
else:
|
||||
l += [ token ]
|
||||
|
||||
print(l)
|
||||
```
|
||||
|
||||
Which should result in the following (though less cleanly formated):
|
||||
|
||||
```python
|
||||
[ { 'word': 'Engelbert', 'score': 0.99..., 'entity': 'PRS'},
|
||||
{ 'word': 'Volvon', 'score': 0.99..., 'entity': 'OBJ'},
|
||||
{ 'word': 'Tele2', 'score': 0.99..., 'entity': 'LOC'},
|
||||
{ 'word': 'Arena', 'score': 0.99..., 'entity': 'LOC'},
|
||||
{ 'word': 'Djurgården', 'score': 0.99..., 'entity': 'ORG'},
|
||||
{ 'word': 'IF', 'score': 0.99..., 'entity': 'ORG'},
|
||||
{ 'word': 'VM', 'score': 0.99..., 'entity': 'EVN'},
|
||||
{ 'word': 'klockan', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'två', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'på', 'score': 0.99..., 'entity': 'TME'},
|
||||
{ 'word': 'kvällen', 'score': 0.54..., 'entity': 'TME'} ]
|
||||
```
|
||||
|
||||
### ALBERT base
|
||||
|
||||
The easisest way to do this is, again, using Huggingface Transformers:
|
||||
|
||||
```python
|
||||
from transformers import AutoModel,AutoTokenizer
|
||||
|
||||
tok = AutoTokenizer.from_pretrained('KB/albert-base-swedish-cased-alpha'),
|
||||
model = AutoModel.from_pretrained('KB/albert-base-swedish-cased-alpha')
|
||||
```
|
||||
|
||||
## Acknowledgements ❤️
|
||||
|
||||
- Resources from Stockholms University, Umeå University and Swedish Language Bank at Gothenburg University were used when fine-tuning BERT for NER.
|
||||
- Model pretraining was made partly in-house at the KBLab and partly (for material without active copyright) with the support of Cloud TPUs from Google's TensorFlow Research Cloud (TFRC).
|
||||
- Models are hosted on S3 by Huggingface 🤗
|
||||
|
||||
@@ -1,141 +0,0 @@
|
||||
---
|
||||
language: it
|
||||
---
|
||||
|
||||
# GePpeTto GPT2 Model 🇮🇹
|
||||
|
||||
Pretrained GPT2 117M model for Italian.
|
||||
|
||||
You can find further details in the paper:
|
||||
|
||||
Lorenzo De Mattei, Michele Cafagna, Felice Dell’Orletta, Malvina Nissim, Marco Guerini "GePpeTto Carves Italian into a Language Model", arXiv preprint. Pdf available at: https://arxiv.org/abs/2004.14253
|
||||
|
||||
## Pretraining Corpus
|
||||
|
||||
The pretraining set comprises two main sources. The first one is a dump of Italian Wikipedia (November 2019),
|
||||
consisting of 2.8GB of text. The second one is the ItWac corpus (Baroni et al., 2009), which amounts to 11GB of web
|
||||
texts. This collection provides a mix of standard and less standard Italian, on a rather wide chronological span,
|
||||
with older texts than the Wikipedia dump (the latter stretches only to the late 2000s).
|
||||
|
||||
## Pretraining details
|
||||
|
||||
This model was trained using GPT2's Hugging Face implemenation on 4 NVIDIA Tesla T4 GPU for 620k steps.
|
||||
|
||||
Training parameters:
|
||||
|
||||
- GPT-2 small configuration
|
||||
- vocabulary size: 30k
|
||||
- Batch size: 32
|
||||
- Block size: 100
|
||||
- Adam Optimizer
|
||||
- Initial learning rate: 5e-5
|
||||
- Warm up steps: 10k
|
||||
|
||||
## Perplexity scores
|
||||
|
||||
| Domain | Perplexity |
|
||||
|---|---|
|
||||
| Wikipedia | 26.1052 |
|
||||
| ItWac | 30.3965 |
|
||||
| Legal | 37.2197 |
|
||||
| News | 45.3859 |
|
||||
| Social Media | 84.6408 |
|
||||
|
||||
For further details, qualitative analysis and human evaluation check out: https://arxiv.org/abs/2004.14253
|
||||
|
||||
## Load Pretrained Model
|
||||
|
||||
You can use this model by installing Huggingface library `transformers`. And you can use it directly by initializing it like this:
|
||||
|
||||
```python
|
||||
from transformers import GPT2Tokenizer, GPT2Model
|
||||
|
||||
model = GPT2Model.from_pretrained('LorenzoDeMattei/GePpeTto')
|
||||
tokenizer = GPT2Tokenizer.from_pretrained(
|
||||
'LorenzoDeMattei/GePpeTto',
|
||||
)
|
||||
```
|
||||
|
||||
## Example using GPT2LMHeadModel
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModelWithLMHead, pipeline, GPT2Tokenizer
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("LorenzoDeMattei/GePpeTto")
|
||||
model = AutoModelWithLMHead.from_pretrained("LorenzoDeMattei/GePpeTto")
|
||||
|
||||
text_generator = pipeline('text-generation', model=model, tokenizer=tokenizer)
|
||||
prompts = [
|
||||
"Wikipedia Geppetto",
|
||||
"Maestro Ciliegia regala il pezzo di legno al suo amico Geppetto, il quale lo prende per fabbricarsi un burattino maraviglioso"]
|
||||
|
||||
|
||||
samples_outputs = text_generator(
|
||||
prompts,
|
||||
do_sample=True,
|
||||
max_length=50,
|
||||
top_k=50,
|
||||
top_p=0.95,
|
||||
num_return_sequences=3
|
||||
)
|
||||
|
||||
|
||||
for i, sample_outputs in enumerate(samples_outputs):
|
||||
print(100 * '-')
|
||||
print("Prompt:", prompts[i])
|
||||
for sample_output in sample_outputs:
|
||||
print("Sample:", sample_output['generated_text'])
|
||||
print()
|
||||
|
||||
```
|
||||
|
||||
Output is,
|
||||
|
||||
```
|
||||
----------------------------------------------------------------------------------------------------
|
||||
Prompt: Wikipedia Geppetto
|
||||
Sample: Wikipedia Geppetto rosso (film 1920)
|
||||
|
||||
Geppetto rosso ("The Smokes in the Black") è un film muto del 1920 diretto da Henry H. Leonard.
|
||||
|
||||
Il film fu prodotto dalla Selig Poly
|
||||
|
||||
Sample: Wikipedia Geppetto
|
||||
|
||||
Geppetto ("Geppetto" in piemontese) è un comune italiano di 978 abitanti della provincia di Cuneo in Piemonte.
|
||||
|
||||
L'abitato, che si trova nel versante valtellinese, si sviluppa nella
|
||||
|
||||
Sample: Wikipedia Geppetto di Natale (romanzo)
|
||||
|
||||
Geppetto di Natale è un romanzo di Mario Caiano, pubblicato nel 2012.
|
||||
|
||||
----------------------------------------------------------------------------------------------------
|
||||
Prompt: Maestro Ciliegia regala il pezzo di legno al suo amico Geppetto, il quale lo prende per fabbricarsi un burattino maraviglioso
|
||||
Sample: Maestro Ciliegia regala il pezzo di legno al suo amico Geppetto, il quale lo prende per fabbricarsi un burattino maraviglioso. Il burattino riesce a scappare. Dopo aver trovato un prezioso sacchetto si reca
|
||||
|
||||
Sample: Maestro Ciliegia regala il pezzo di legno al suo amico Geppetto, il quale lo prende per fabbricarsi un burattino maraviglioso, e l'unico che lo possiede, ma, di fronte a tutte queste prove
|
||||
|
||||
Sample: Maestro Ciliegia regala il pezzo di legno al suo amico Geppetto, il quale lo prende per fabbricarsi un burattino maraviglioso: - A voi gli occhi, le guance! A voi il mio pezzo!
|
||||
```
|
||||
|
||||
## Citation
|
||||
|
||||
Please use the following bibtex entry:
|
||||
|
||||
```
|
||||
@misc{mattei2020geppetto,
|
||||
title={GePpeTto Carves Italian into a Language Model},
|
||||
author={Lorenzo De Mattei and Michele Cafagna and Felice Dell'Orletta and Malvina Nissim and Marco Guerini},
|
||||
year={2020},
|
||||
eprint={2004.14253},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL}
|
||||
}
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
Marco Baroni, Silvia Bernardini, Adriano Ferraresi,
|
||||
and Eros Zanchetta. 2009. The WaCky wide web: a
|
||||
collection of very large linguistically processed webcrawled corpora. Language resources and evaluation, 43(3):209–226.
|
||||
@@ -1,47 +0,0 @@
|
||||
## About the model
|
||||
|
||||
The model has been trained on a collection of 500k articles with headings. Its purpose is to create a one-line heading suitable for the given article.
|
||||
|
||||
Sample code with a WikiNews article:
|
||||
|
||||
```python
|
||||
import torch
|
||||
from transformers import T5ForConditionalGeneration,T5Tokenizer
|
||||
|
||||
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
||||
|
||||
model = T5ForConditionalGeneration.from_pretrained("Michau/t5-base-en-generate-headline")
|
||||
tokenizer = T5Tokenizer.from_pretrained("Michau/t5-base-en-generate-headline")
|
||||
model = model.to(device)
|
||||
|
||||
article = '''
|
||||
Very early yesterday morning, the United States President Donald Trump reported he and his wife First Lady Melania Trump tested positive for COVID-19. Officials said the Trumps' 14-year-old son Barron tested negative as did First Family and Senior Advisors Jared Kushner and Ivanka Trump.
|
||||
Trump took to social media, posting at 12:54 am local time (0454 UTC) on Twitter, "Tonight, [Melania] and I tested positive for COVID-19. We will begin our quarantine and recovery process immediately. We will get through this TOGETHER!" Yesterday afternoon Marine One landed on the White House's South Lawn flying Trump to Walter Reed National Military Medical Center (WRNMMC) in Bethesda, Maryland.
|
||||
Reports said both were showing "mild symptoms". Senior administration officials were tested as people were informed of the positive test. Senior advisor Hope Hicks had tested positive on Thursday.
|
||||
Presidential physician Sean Conley issued a statement saying Trump has been given zinc, vitamin D, Pepcid and a daily Aspirin. Conley also gave a single dose of the experimental polyclonal antibodies drug from Regeneron Pharmaceuticals.
|
||||
According to official statements, Trump, now operating from the WRNMMC, is to continue performing his duties as president during a 14-day quarantine. In the event of Trump becoming incapacitated, Vice President Mike Pence could take over the duties of president via the 25th Amendment of the US Constitution. The Pence family all tested negative as of yesterday and there were no changes regarding Pence's campaign events.
|
||||
'''
|
||||
|
||||
text = "headline: " + article
|
||||
|
||||
max_len = 256
|
||||
|
||||
encoding = tokenizer.encode_plus(text, return_tensors = "pt")
|
||||
input_ids = encoding["input_ids"].to(device)
|
||||
attention_masks = encoding["attention_mask"].to(device)
|
||||
|
||||
beam_outputs = model.generate(
|
||||
input_ids = input_ids,
|
||||
attention_mask = attention_masks,
|
||||
max_length = 64,
|
||||
num_beams = 3,
|
||||
early_stopping = True,
|
||||
)
|
||||
|
||||
result = tokenizer.decode(beam_outputs[0])
|
||||
print(result)
|
||||
```
|
||||
|
||||
Result:
|
||||
|
||||
```Trump and First Lady Melania Test Positive for COVID-19```
|
||||
@@ -1,74 +0,0 @@
|
||||
---
|
||||
language: tn
|
||||
---
|
||||
|
||||
# TswanaBert
|
||||
Pretrained model on the Tswana language using a masked language modeling (MLM) objective.
|
||||
|
||||
## Model Description.
|
||||
TswanaBERT is a transformer model pre-trained on a corpus of Setswana in a self-supervised fashion by masking part of the input words and training to predict the masks by using byte-level tokens.
|
||||
|
||||
## Intended uses & limitations
|
||||
The model can be used for either masked language modeling or next word prediction. It can also be fine-tuned on a specific down-stream NLP application.
|
||||
|
||||
#### How to use
|
||||
|
||||
```python
|
||||
>>> from transformers import pipeline
|
||||
>>> from transformers import AutoTokenizer, AutoModelWithLMHead
|
||||
|
||||
>>> tokenizer = AutoTokenizer.from_pretrained("MoseliMotsoehli/TswanaBert")
|
||||
>>> model = AutoModelWithLMHead.from_pretrained("MoseliMotsoehli/TswanaBert")
|
||||
>>> unmasker = pipeline('fill-mask', model=model, tokenizer=tokenizer)
|
||||
>>> unmasker("Ntshopotse <mask> e godile.")
|
||||
|
||||
[{'score': 0.32749542593955994,
|
||||
'sequence': '<s>Ntshopotse setse e godile.</s>',
|
||||
'token': 538,
|
||||
'token_str': 'Ġsetse'},
|
||||
{'score': 0.060260992497205734,
|
||||
'sequence': '<s>Ntshopotse le e godile.</s>',
|
||||
'token': 270,
|
||||
'token_str': 'Ġle'},
|
||||
{'score': 0.058460816740989685,
|
||||
'sequence': '<s>Ntshopotse bone e godile.</s>',
|
||||
'token': 364,
|
||||
'token_str': 'Ġbone'},
|
||||
{'score': 0.05694682151079178,
|
||||
'sequence': '<s>Ntshopotse ga e godile.</s>',
|
||||
'token': 298,
|
||||
'token_str': 'Ġga'},
|
||||
{'score': 0.0565204992890358,
|
||||
'sequence': '<s>Ntshopotse, e godile.</s>',
|
||||
'token': 16,
|
||||
'token_str': ','}]
|
||||
```
|
||||
|
||||
#### Limitations and bias
|
||||
The model is trained on a relatively small collection of setwana, mostly from news articles and creative writtings, and so is not representative enough of the language as yet.
|
||||
|
||||
## Training data
|
||||
|
||||
1. The largest portion of this dataset (10k) sentences of text, comes from the [Leipzig Corpora Collection](https://wortschatz.uni-leipzig.de/en/download)
|
||||
|
||||
2. I Then added SABC news headlines collected by Marivate Vukosi, & Sefara Tshephisho, (2020) that is generously made available on [zenoodo](http://doi.org/10.5281/zenodo.3668495 ). This added 185 tswana sentences to my corpus.
|
||||
|
||||
3. I went on to add 300 more sentences by scrapping following news sites and blogs that mosty originate in Botswana. I actively continue to expand the dataset.
|
||||
|
||||
* http://setswana.blogspot.com/
|
||||
* https://omniglot.com/writing/tswana.php
|
||||
* http://www.dailynews.gov.bw/
|
||||
* http://www.mmegi.bw/index.php
|
||||
* https://tsena.co.bw
|
||||
* http://www.botswana.co.za/Cultural_Issues-travel/botswana-country-guide-en-route.html
|
||||
* https://www.poemhunter.com/poem/2013-setswana/
|
||||
https://www.poemhunter.com/poem/ngwana-wa-mosetsana/
|
||||
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{author = {Moseli Motsoehli},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
@@ -1,56 +0,0 @@
|
||||
---
|
||||
language: zu
|
||||
---
|
||||
|
||||
# zuBERTa
|
||||
zuBERTa is a RoBERTa style transformer language model trained on zulu text.
|
||||
|
||||
## Intended uses & limitations
|
||||
The model can be used for getting embeddings to use on a down-stream task such as question answering.
|
||||
|
||||
#### How to use
|
||||
|
||||
```python
|
||||
>>> from transformers import pipeline
|
||||
>>> from transformers import AutoTokenizer, AutoModelWithLMHead
|
||||
|
||||
>>> tokenizer = AutoTokenizer.from_pretrained("MoseliMotsoehli/zuBERTa")
|
||||
>>> model = AutoModelWithLMHead.from_pretrained("MoseliMotsoehli/zuBERTa")
|
||||
>>> unmasker = pipeline('fill-mask', model=model, tokenizer=tokenizer)
|
||||
>>> unmasker("Abafika eNkandla bafika sebeholwa <mask> uMpongo kaZingelwayo.")
|
||||
|
||||
[
|
||||
{
|
||||
"sequence": "<s>Abafika eNkandla bafika sebeholwa khona uMpongo kaZingelwayo.</s>",
|
||||
"score": 0.050459690392017365,
|
||||
"token": 555,
|
||||
"token_str": "Ġkhona"
|
||||
},
|
||||
{
|
||||
"sequence": "<s>Abafika eNkandla bafika sebeholwa inkosi uMpongo kaZingelwayo.</s>",
|
||||
"score": 0.03668094798922539,
|
||||
"token": 2321,
|
||||
"token_str": "Ġinkosi"
|
||||
},
|
||||
{
|
||||
"sequence": "<s>Abafika eNkandla bafika sebeholwa ubukhosi uMpongo kaZingelwayo.</s>",
|
||||
"score": 0.028774697333574295,
|
||||
"token": 5101,
|
||||
"token_str": "Ġubukhosi"
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
## Training data
|
||||
|
||||
1. 30k sentences of text, came from the [Leipzig Corpora Collection](https://wortschatz.uni-leipzig.de/en/download) of zulu 2018. These were collected from news articles and creative writtings.
|
||||
2. ~7500 articles of human generated translations were scraped from the zulu [wikipedia](https://zu.wikipedia.org/wiki/Special:AllPages).
|
||||
|
||||
### BibTeX entry and citation info
|
||||
|
||||
```bibtex
|
||||
@inproceedings{author = {Moseli Motsoehli},
|
||||
title = {Towards transformation of Southern African language models through transformers.},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
@@ -1,118 +0,0 @@
|
||||
---
|
||||
language: it
|
||||
---
|
||||
|
||||
# UmBERTo Commoncrawl Cased
|
||||
|
||||
[UmBERTo](https://github.com/musixmatchresearch/umberto) is a Roberta-based Language Model trained on large Italian Corpora and uses two innovative approaches: SentencePiece and Whole Word Masking. Now available at [github.com/huggingface/transformers](https://huggingface.co/Musixmatch/umberto-commoncrawl-cased-v1)
|
||||
|
||||
<p align="center">
|
||||
<img src="https://user-images.githubusercontent.com/7140210/72913702-d55a8480-3d3d-11ea-99fc-f2ef29af4e72.jpg" width="700"> </br>
|
||||
Marco Lodola, Monument to Umberto Eco, Alessandria 2019
|
||||
</p>
|
||||
|
||||
## Dataset
|
||||
UmBERTo-Commoncrawl-Cased utilizes the Italian subcorpus of [OSCAR](https://traces1.inria.fr/oscar/) as training set of the language model. We used deduplicated version of the Italian corpus that consists in 70 GB of plain text data, 210M sentences with 11B words where the sentences have been filtered and shuffled at line level in order to be used for NLP research.
|
||||
|
||||
## Pre-trained model
|
||||
|
||||
| Model | WWM | Cased | Tokenizer | Vocab Size | Train Steps | Download |
|
||||
| ------ | ------ | ------ | ------ | ------ |------ | ------ |
|
||||
| `umberto-commoncrawl-cased-v1` | YES | YES | SPM | 32K | 125k | [Link](http://bit.ly/35zO7GH) |
|
||||
|
||||
This model was trained with [SentencePiece](https://github.com/google/sentencepiece) and Whole Word Masking.
|
||||
|
||||
## Downstream Tasks
|
||||
These results refers to umberto-commoncrawl-cased model. All details are at [Umberto](https://github.com/musixmatchresearch/umberto) Official Page.
|
||||
|
||||
#### Named Entity Recognition (NER)
|
||||
|
||||
| Dataset | F1 | Precision | Recall | Accuracy |
|
||||
| ------ | ------ | ------ | ------ | ------ |
|
||||
| **ICAB-EvalITA07** | **87.565** | 86.596 | 88.556 | 98.690 |
|
||||
| **WikiNER-ITA** | **92.531** | 92.509 | 92.553 | 99.136 |
|
||||
|
||||
#### Part of Speech (POS)
|
||||
|
||||
| Dataset | F1 | Precision | Recall | Accuracy |
|
||||
| ------ | ------ | ------ | ------ | ------ |
|
||||
| **UD_Italian-ISDT** | 98.870 | 98.861 | 98.879 | **98.977** |
|
||||
| **UD_Italian-ParTUT** | 98.786 | 98.812 | 98.760 | **98.903** |
|
||||
|
||||
|
||||
|
||||
## Usage
|
||||
|
||||
##### Load UmBERTo with AutoModel, Autotokenizer:
|
||||
|
||||
```python
|
||||
|
||||
import torch
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Musixmatch/umberto-commoncrawl-cased-v1")
|
||||
umberto = AutoModel.from_pretrained("Musixmatch/umberto-commoncrawl-cased-v1")
|
||||
|
||||
encoded_input = tokenizer.encode("Umberto Eco è stato un grande scrittore")
|
||||
input_ids = torch.tensor(encoded_input).unsqueeze(0) # Batch size 1
|
||||
outputs = umberto(input_ids)
|
||||
last_hidden_states = outputs[0] # The last hidden-state is the first element of the output
|
||||
```
|
||||
|
||||
##### Predict masked token:
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
fill_mask = pipeline(
|
||||
"fill-mask",
|
||||
model="Musixmatch/umberto-commoncrawl-cased-v1",
|
||||
tokenizer="Musixmatch/umberto-commoncrawl-cased-v1"
|
||||
)
|
||||
|
||||
result = fill_mask("Umberto Eco è <mask> un grande scrittore")
|
||||
# {'sequence': '<s> Umberto Eco è considerato un grande scrittore</s>', 'score': 0.18599839508533478, 'token': 5032}
|
||||
# {'sequence': '<s> Umberto Eco è stato un grande scrittore</s>', 'score': 0.17816807329654694, 'token': 471}
|
||||
# {'sequence': '<s> Umberto Eco è sicuramente un grande scrittore</s>', 'score': 0.16565583646297455, 'token': 2654}
|
||||
# {'sequence': '<s> Umberto Eco è indubbiamente un grande scrittore</s>', 'score': 0.0932890921831131, 'token': 17908}
|
||||
# {'sequence': '<s> Umberto Eco è certamente un grande scrittore</s>', 'score': 0.054701317101716995, 'token': 5269}
|
||||
```
|
||||
|
||||
|
||||
## Citation
|
||||
All of the original datasets are publicly available or were released with the owners' grant. The datasets are all released under a CC0 or CCBY license.
|
||||
|
||||
* UD Italian-ISDT Dataset [Github](https://github.com/UniversalDependencies/UD_Italian-ISDT)
|
||||
* UD Italian-ParTUT Dataset [Github](https://github.com/UniversalDependencies/UD_Italian-ParTUT)
|
||||
* I-CAB (Italian Content Annotation Bank), EvalITA [Page](http://www.evalita.it/)
|
||||
* WIKINER [Page](https://figshare.com/articles/Learning_multilingual_named_entity_recognition_from_Wikipedia/5462500) , [Paper](https://www.sciencedirect.com/science/article/pii/S0004370212000276?via%3Dihub)
|
||||
|
||||
```
|
||||
@inproceedings {magnini2006annotazione,
|
||||
title = {Annotazione di contenuti concettuali in un corpus italiano: I - CAB},
|
||||
author = {Magnini,Bernardo and Cappelli,Amedeo and Pianta,Emanuele and Speranza,Manuela and Bartalesi Lenzi,V and Sprugnoli,Rachele and Romano,Lorenza and Girardi,Christian and Negri,Matteo},
|
||||
booktitle = {Proc.of SILFI 2006},
|
||||
year = {2006}
|
||||
}
|
||||
@inproceedings {magnini2006cab,
|
||||
title = {I - CAB: the Italian Content Annotation Bank.},
|
||||
author = {Magnini,Bernardo and Pianta,Emanuele and Girardi,Christian and Negri,Matteo and Romano,Lorenza and Speranza,Manuela and Lenzi,Valentina Bartalesi and Sprugnoli,Rachele},
|
||||
booktitle = {LREC},
|
||||
pages = {963--968},
|
||||
year = {2006},
|
||||
organization = {Citeseer}
|
||||
}
|
||||
```
|
||||
|
||||
## Authors
|
||||
|
||||
**Loreto Parisi**: `loreto at musixmatch dot com`, [loretoparisi](https://github.com/loretoparisi)
|
||||
**Simone Francia**: `simone.francia at musixmatch dot com`, [simonefrancia](https://github.com/simonefrancia)
|
||||
**Paolo Magnani**: `paul.magnani95 at gmail dot com`, [paulthemagno](https://github.com/paulthemagno)
|
||||
|
||||
## About Musixmatch AI
|
||||

|
||||
We do Machine Learning and Artificial Intelligence @[musixmatch](https://twitter.com/Musixmatch)
|
||||
Follow us on [Twitter](https://twitter.com/musixmatchai) [Github](https://github.com/musixmatchresearch)
|
||||
|
||||
|
||||
@@ -1,117 +0,0 @@
|
||||
---
|
||||
language: it
|
||||
---
|
||||
|
||||
# UmBERTo Wikipedia Uncased
|
||||
|
||||
[UmBERTo](https://github.com/musixmatchresearch/umberto) is a Roberta-based Language Model trained on large Italian Corpora and uses two innovative approaches: SentencePiece and Whole Word Masking. Now available at [github.com/huggingface/transformers](https://huggingface.co/Musixmatch/umberto-commoncrawl-cased-v1)
|
||||
|
||||
<p align="center">
|
||||
<img src="https://user-images.githubusercontent.com/7140210/72913702-d55a8480-3d3d-11ea-99fc-f2ef29af4e72.jpg" width="700"> </br>
|
||||
Marco Lodola, Monument to Umberto Eco, Alessandria 2019
|
||||
</p>
|
||||
|
||||
## Dataset
|
||||
UmBERTo-Wikipedia-Uncased Training is trained on a relative small corpus (~7GB) extracted from [Wikipedia-ITA](https://linguatools.org/tools/corpora/wikipedia-monolingual-corpora/).
|
||||
|
||||
## Pre-trained model
|
||||
|
||||
| Model | WWM | Cased | Tokenizer | Vocab Size | Train Steps | Download |
|
||||
| ------ | ------ | ------ | ------ | ------ |------ | ------ |
|
||||
| `umberto-wikipedia-uncased-v1` | YES | YES | SPM | 32K | 100k | [Link](http://bit.ly/35wbSj6) |
|
||||
|
||||
This model was trained with [SentencePiece](https://github.com/google/sentencepiece) and Whole Word Masking.
|
||||
|
||||
## Downstream Tasks
|
||||
These results refers to umberto-wikipedia-uncased model. All details are at [Umberto](https://github.com/musixmatchresearch/umberto) Official Page.
|
||||
|
||||
#### Named Entity Recognition (NER)
|
||||
|
||||
| Dataset | F1 | Precision | Recall | Accuracy |
|
||||
| ------ | ------ | ------ | ------ | ----- |
|
||||
| **ICAB-EvalITA07** | **86.240** | 85.939 | 86.544 | 98.534 |
|
||||
| **WikiNER-ITA** | **90.483** | 90.328 | 90.638 | 98.661 |
|
||||
|
||||
#### Part of Speech (POS)
|
||||
|
||||
| Dataset | F1 | Precision | Recall | Accuracy |
|
||||
| ------ | ------ | ------ | ------ | ------ |
|
||||
| **UD_Italian-ISDT** | 98.563 | 98.508 | 98.618 | **98.717** |
|
||||
| **UD_Italian-ParTUT** | 97.810 | 97.835 | 97.784 | **98.060** |
|
||||
|
||||
|
||||
|
||||
## Usage
|
||||
|
||||
##### Load UmBERTo Wikipedia Uncased with AutoModel, Autotokenizer:
|
||||
|
||||
```python
|
||||
|
||||
import torch
|
||||
from transformers import AutoTokenizer, AutoModel
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Musixmatch/umberto-wikipedia-uncased-v1")
|
||||
umberto = AutoModel.from_pretrained("Musixmatch/umberto-wikipedia-uncased-v1")
|
||||
|
||||
encoded_input = tokenizer.encode("Umberto Eco è stato un grande scrittore")
|
||||
input_ids = torch.tensor(encoded_input).unsqueeze(0) # Batch size 1
|
||||
outputs = umberto(input_ids)
|
||||
last_hidden_states = outputs[0] # The last hidden-state is the first element of the output
|
||||
```
|
||||
|
||||
##### Predict masked token:
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
fill_mask = pipeline(
|
||||
"fill-mask",
|
||||
model="Musixmatch/umberto-wikipedia-uncased-v1",
|
||||
tokenizer="Musixmatch/umberto-wikipedia-uncased-v1"
|
||||
)
|
||||
|
||||
result = fill_mask("Umberto Eco è <mask> un grande scrittore")
|
||||
# {'sequence': '<s> umberto eco è stato un grande scrittore</s>', 'score': 0.5784581303596497, 'token': 361}
|
||||
# {'sequence': '<s> umberto eco è anche un grande scrittore</s>', 'score': 0.33813193440437317, 'token': 269}
|
||||
# {'sequence': '<s> umberto eco è considerato un grande scrittore</s>', 'score': 0.027196012437343597, 'token': 3236}
|
||||
# {'sequence': '<s> umberto eco è diventato un grande scrittore</s>', 'score': 0.013716378249228, 'token': 5742}
|
||||
# {'sequence': '<s> umberto eco è inoltre un grande scrittore</s>', 'score': 0.010662357322871685, 'token': 1030}
|
||||
```
|
||||
|
||||
|
||||
## Citation
|
||||
All of the original datasets are publicly available or were released with the owners' grant. The datasets are all released under a CC0 or CCBY license.
|
||||
|
||||
* UD Italian-ISDT Dataset [Github](https://github.com/UniversalDependencies/UD_Italian-ISDT)
|
||||
* UD Italian-ParTUT Dataset [Github](https://github.com/UniversalDependencies/UD_Italian-ParTUT)
|
||||
* I-CAB (Italian Content Annotation Bank), EvalITA [Page](http://www.evalita.it/)
|
||||
* WIKINER [Page](https://figshare.com/articles/Learning_multilingual_named_entity_recognition_from_Wikipedia/5462500) , [Paper](https://www.sciencedirect.com/science/article/pii/S0004370212000276?via%3Dihub)
|
||||
|
||||
```
|
||||
@inproceedings {magnini2006annotazione,
|
||||
title = {Annotazione di contenuti concettuali in un corpus italiano: I - CAB},
|
||||
author = {Magnini,Bernardo and Cappelli,Amedeo and Pianta,Emanuele and Speranza,Manuela and Bartalesi Lenzi,V and Sprugnoli,Rachele and Romano,Lorenza and Girardi,Christian and Negri,Matteo},
|
||||
booktitle = {Proc.of SILFI 2006},
|
||||
year = {2006}
|
||||
}
|
||||
@inproceedings {magnini2006cab,
|
||||
title = {I - CAB: the Italian Content Annotation Bank.},
|
||||
author = {Magnini,Bernardo and Pianta,Emanuele and Girardi,Christian and Negri,Matteo and Romano,Lorenza and Speranza,Manuela and Lenzi,Valentina Bartalesi and Sprugnoli,Rachele},
|
||||
booktitle = {LREC},
|
||||
pages = {963--968},
|
||||
year = {2006},
|
||||
organization = {Citeseer}
|
||||
}
|
||||
```
|
||||
|
||||
## Authors
|
||||
|
||||
**Loreto Parisi**: `loreto at musixmatch dot com`, [loretoparisi](https://github.com/loretoparisi)
|
||||
**Simone Francia**: `simone.francia at musixmatch dot com`, [simonefrancia](https://github.com/simonefrancia)
|
||||
**Paolo Magnani**: `paul.magnani95 at gmail dot com`, [paulthemagno](https://github.com/paulthemagno)
|
||||
|
||||
## About Musixmatch AI
|
||||

|
||||
We do Machine Learning and Artificial Intelligence @[musixmatch](https://twitter.com/Musixmatch)
|
||||
Follow us on [Twitter](https://twitter.com/musixmatchai) [Github](https://github.com/musixmatchresearch)
|
||||
|
||||
@@ -1,54 +0,0 @@
|
||||
# MS-BERT
|
||||
|
||||
## Introduction
|
||||
|
||||
This repository provides codes and models of MS-BERT.
|
||||
MS-BERT was pre-trained on notes from neurological examination for Multiple Sclerosis (MS) patients at St. Michael's Hospital in Toronto, Canada.
|
||||
|
||||
## Data
|
||||
|
||||
The dataset contained approximately 75,000 clinical notes, for about 5000 patients, totaling to over 35.7 million words.
|
||||
These notes were collected from patients who visited St. Michael's Hospital MS Clinic between 2015 to 2019.
|
||||
The notes contained a variety of information pertaining to a neurological exam.
|
||||
For example, a note can contain information on the patient's condition, their progress over time and diagnosis.
|
||||
The gender split within the dataset was observed to be 72% female and 28% male ([which reflects the natural discrepancy seen in MS][1]).
|
||||
Further sections will describe how MS-BERT was pre trained through the use of these clinically relevant and rich neurological notes.
|
||||
|
||||
## Data pre-processing
|
||||
|
||||
The data was pre-processed to remove any identifying information. This includes information on: patient names, doctor names, hospital names, patient identification numbers, phone numbers, addresses, and time. In order to de-identify the information, we used a curated database that contained patient and doctor information. This curated database was paired with regular expressions to find and remove any identifying pieces of information. Each of these identifiers were replaced with a specific token. These tokens were chosen based on three criteria: (1) they belong to the current BERT vocab, (2), they have relatively the same semantic meaning as the word they are replacing, and (3), the token is not found in the original unprocessed dataset. The replacements that met the criteria above were as follows:
|
||||
|
||||
Female first names -> Lucie
|
||||
|
||||
Male first names -> Ezekiel
|
||||
|
||||
Last/family names -> Salamanca.
|
||||
|
||||
Dates -> 2010s
|
||||
|
||||
Patient IDs -> 999
|
||||
|
||||
Phone numbers -> 1718
|
||||
|
||||
Addresses -> Silesia
|
||||
|
||||
Time -> 1610
|
||||
|
||||
Locations/Hospital/Clinic names -> Troy
|
||||
|
||||
## Pre-training
|
||||
|
||||
The starting point for our model is the already pre-trained and fine-tuned BLUE-BERT base. We further pre-train it using the masked language modelling task from the huggingface transformers [library](https://github.com/huggingface).
|
||||
|
||||
The hyperparameters can be found in the config file in this repository or [here](https://s3.amazonaws.com/models.huggingface.co/bert/NLP4H/ms_bert/config.json)
|
||||
|
||||
## Acknowledgements
|
||||
|
||||
We would like to thank the researchers and staff at the Data Science and Advanced Analytics (DSAA) department, St. Michael’s Hospital, for providing consistent support and guidance throughout this project.
|
||||
We would also like to thank Dr. Marzyeh Ghassemi, Taylor Killan, Nathan Ng and Haoran Zhang for providing us the opportunity to work on this exciting project.
|
||||
|
||||
## Disclaimer
|
||||
|
||||
MS-BERT shows the results of research conducted at the Data Science and Advanced Analytics (DSAA) department, St. Michael’s Hospital. The results produced by MS-BERT are not intended for direct diagnostic use or medical decision-making without review and oversight by a clinical professional. Individuals should not make decisions about their health solely on the basis of the results produced by MS-BERT. St. Michael’s Hospital does not independently verify the validity or utility of the results produced by MS-BERT. If you have questions about the results produced by MS-BERT please consult a healthcare professional. If you would like more information about the research conducted at DSAA please contact [Zhen Yang](mailto:zhen.yang@unityhealth.to). If you would like more information on neurological examination notes please contact [Dr. Tony Antoniou](mailto:tony.antoniou@unityhealth.to) or [Dr. Jiwon Oh](mailto:jiwon.oh@unityhealth.to) from the MS clinic at St. Michael's Hospital.
|
||||
|
||||
[1]: https://www.ncbi.nlm.nih.gov/pmc/articles/PMC3707353/
|
||||
@@ -1,28 +0,0 @@
|
||||
---
|
||||
language: kn
|
||||
---
|
||||
|
||||
# Welcome to KanBERTo (ಕನ್ಬರ್ಟೋ)
|
||||
|
||||
## Model Description
|
||||
|
||||
> This is a small language model for [Kannada](https://en.wikipedia.org/wiki/Kannada) language with 1M data samples taken from
|
||||
[OSCAR page](https://traces1.inria.fr/oscar/files/compressed-orig/kn.txt.gz)
|
||||
|
||||
## Training params
|
||||
|
||||
- **Dataset** - 1M data samples are used to train this model from OSCAR page(https://traces1.inria.fr/oscar/) eventhough data set is of 1.7 GB due to resource constraint to train
|
||||
I have picked only 1M data from the total 1.7GB data set. If you are interested in collaboration and have computational resources to train on you are most welcome to do so.
|
||||
|
||||
- **Preprocessing** - ByteLevelBPETokenizer is used to tokenize the sentences at character level and vocabulary size is set to 52k as per standard values given by 🤗
|
||||
- **Hyperparameters** - __ByteLevelBPETokenizer__ : vocabulary size = 52_000 and min_frequency = 2
|
||||
__Trainer__ : num_train_epochs=12 - trained for 12 epochs
|
||||
per_gpu_train_batch_size=64 - batch size for the datasamples is 64
|
||||
save_steps=10_000 - save model for every 10k steps
|
||||
save_total_limit=2 - save limit is set for 2
|
||||
|
||||
**Intended uses & limitations**
|
||||
this is for anyone who wants to make use of kannada language models for various tasks like language generation, translation and many more use cases.
|
||||
|
||||
**Whatever else is helpful!**
|
||||
If you are intersted in collaboration feel free to reach me [Naveen](mailto:naveen.maltesh@gmail.com)
|
||||
@@ -1,26 +0,0 @@
|
||||
# BERT-Small CORD-19 fine-tuned on SQuAD 2.0
|
||||
|
||||
[bert-small-cord19 model](https://huggingface.co/NeuML/bert-small-cord19) fine-tuned on SQuAD 2.0
|
||||
|
||||
## Building the model
|
||||
|
||||
```bash
|
||||
python run_squad.py
|
||||
--model_type bert
|
||||
--model_name_or_path bert-small-cord19
|
||||
--do_train
|
||||
--do_eval
|
||||
--do_lower_case
|
||||
--version_2_with_negative
|
||||
--train_file train-v2.0.json
|
||||
--predict_file dev-v2.0.json
|
||||
--per_gpu_train_batch_size 8
|
||||
--learning_rate 3e-5
|
||||
--num_train_epochs 3.0
|
||||
--max_seq_length 384
|
||||
--doc_stride 128
|
||||
--output_dir bert-small-cord19-squad2
|
||||
--save_steps 0
|
||||
--threads 8
|
||||
--overwrite_cache
|
||||
--overwrite_output_dir
|
||||
@@ -1,25 +0,0 @@
|
||||
# BERT-Small fine-tuned on CORD-19 dataset
|
||||
|
||||
[BERT L6_H-512_A-8 model](https://huggingface.co/google/bert_uncased_L-6_H-512_A-8) fine-tuned on the [CORD-19 dataset](https://www.semanticscholar.org/cord19).
|
||||
|
||||
## CORD-19 data subset
|
||||
The training data for this dataset is stored as a [Kaggle dataset](https://www.kaggle.com/davidmezzetti/cord19-qa?select=cord19.txt). The training
|
||||
data is a subset of the full corpus, focusing on high-quality, study-design detected articles.
|
||||
|
||||
## Building the model
|
||||
|
||||
```bash
|
||||
python run_language_modeling.py
|
||||
--model_type bert
|
||||
--model_name_or_path google/bert_uncased_L-6_H-512_A-8
|
||||
--do_train
|
||||
--mlm
|
||||
--line_by_line
|
||||
--block_size 512
|
||||
--train_data_file cord19.txt
|
||||
--per_gpu_train_batch_size 4
|
||||
--learning_rate 3e-5
|
||||
--num_train_epochs 3.0
|
||||
--output_dir bert-small-cord19
|
||||
--save_steps 0
|
||||
--overwrite_output_dir
|
||||
@@ -1,63 +0,0 @@
|
||||
# BERT-Small fine-tuned on CORD-19 QA dataset
|
||||
|
||||
[bert-small-cord19-squad model](https://huggingface.co/NeuML/bert-small-cord19-squad2) fine-tuned on the [CORD-19 QA dataset](https://www.kaggle.com/davidmezzetti/cord19-qa?select=cord19-qa.json).
|
||||
|
||||
## CORD-19 QA dataset
|
||||
The CORD-19 QA dataset is a SQuAD 2.0 formatted list of question, context, answer combinations covering the [CORD-19 dataset](https://www.semanticscholar.org/cord19).
|
||||
|
||||
## Building the model
|
||||
|
||||
```bash
|
||||
python run_squad.py \
|
||||
--model_type bert \
|
||||
--model_name_or_path bert-small-cord19-squad \
|
||||
--do_train \
|
||||
--do_lower_case \
|
||||
--version_2_with_negative \
|
||||
--train_file cord19-qa.json \
|
||||
--per_gpu_train_batch_size 8 \
|
||||
--learning_rate 5e-5 \
|
||||
--num_train_epochs 10.0 \
|
||||
--max_seq_length 384 \
|
||||
--doc_stride 128 \
|
||||
--output_dir bert-small-cord19qa \
|
||||
--save_steps 0 \
|
||||
--threads 8 \
|
||||
--overwrite_cache \
|
||||
--overwrite_output_dir
|
||||
```
|
||||
|
||||
## Testing the model
|
||||
|
||||
Example usage below:
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
qa = pipeline(
|
||||
"question-answering",
|
||||
model="NeuML/bert-small-cord19qa",
|
||||
tokenizer="NeuML/bert-small-cord19qa"
|
||||
)
|
||||
|
||||
qa({
|
||||
"question": "What is the median incubation period?",
|
||||
"context": "The incubation period is around 5 days (range: 4-7 days) with a maximum of 12-13 day"
|
||||
})
|
||||
|
||||
qa({
|
||||
"question": "What is the incubation period range?",
|
||||
"context": "The incubation period is around 5 days (range: 4-7 days) with a maximum of 12-13 day"
|
||||
})
|
||||
|
||||
qa({
|
||||
"question": "What type of surfaces does it persist?",
|
||||
"context": "The virus can survive on surfaces for up to 72 hours such as plastic and stainless steel ."
|
||||
})
|
||||
```
|
||||
|
||||
```json
|
||||
{"score": 0.5970273583242793, "start": 32, "end": 38, "answer": "5 days"}
|
||||
{"score": 0.999555868193891, "start": 39, "end": 56, "answer": "(range: 4-7 days)"}
|
||||
{"score": 0.9992726505196998, "start": 61, "end": 88, "answer": "plastic and stainless steel"}
|
||||
```
|
||||
@@ -1,55 +0,0 @@
|
||||
---
|
||||
language: vn
|
||||
---
|
||||
# BERT for Vietnamese is trained on more 20 GB news dataset
|
||||
|
||||
Apply for task sentiment analysis on using [AIViVN's comments dataset](https://www.aivivn.com/contests/6)
|
||||
|
||||
The model achieved 0.90268 on the public leaderboard, (winner's score is 0.90087)
|
||||
Bert4news is used for a toolkit Vietnames(segmentation and Named Entity Recognition) at ViNLPtoolkit(https://github.com/bino282/ViNLP)
|
||||
|
||||
***************New Mar 11 , 2020 ***************
|
||||
|
||||
**[BERT](https://github.com/google-research/bert)** (from Google Research and the Toyota Technological Institute at Chicago) released with the paper [BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding](https://arxiv.org/abs/1810.04805).
|
||||
|
||||
We use word sentencepiece, use basic bert tokenization and same config with bert base with lowercase = False.
|
||||
|
||||
You can download trained model:
|
||||
- [tensorflow](https://drive.google.com/file/d/1X-sRDYf7moS_h61J3L79NkMVGHP-P-k5/view?usp=sharing).
|
||||
- [pytorch](https://drive.google.com/file/d/11aFSTpYIurn-oI2XpAmcCTccB_AonMOu/view?usp=sharing).
|
||||
|
||||
Use with huggingface/transformers
|
||||
``` bash
|
||||
import torch
|
||||
from transformers import AutoTokenizer,AutoModel
|
||||
tokenizer= AutoTokenizer.from_pretrained("NlpHUST/vibert4news-base-cased")
|
||||
bert_model = AutoModel.from_pretrained("NlpHUST/vibert4news-base-cased")
|
||||
|
||||
line = "Tôi là sinh viên trường Bách Khoa Hà Nội ."
|
||||
input_id = tokenizer.encode(line,add_special_tokens = True)
|
||||
att_mask = [int(token_id > 0) for token_id in input_id]
|
||||
input_ids = torch.tensor([input_id])
|
||||
att_masks = torch.tensor([att_mask])
|
||||
with torch.no_grad():
|
||||
features = bert_model(input_ids,att_masks)
|
||||
|
||||
print(features)
|
||||
|
||||
```
|
||||
|
||||
Run training with base config
|
||||
|
||||
``` bash
|
||||
|
||||
python train_pytorch.py \
|
||||
--model_path=bert4news.pytorch \
|
||||
--max_len=200 \
|
||||
--batch_size=16 \
|
||||
--epochs=6 \
|
||||
--lr=2e-5
|
||||
|
||||
```
|
||||
|
||||
### Contact information
|
||||
For personal communication related to this project, please contact Nha Nguyen Van (nha282@gmail.com).
|
||||
|
||||
@@ -1,109 +0,0 @@
|
||||
---
|
||||
language: he
|
||||
|
||||
thumbnail: https://avatars1.githubusercontent.com/u/3617152?norod.jpg
|
||||
widget:
|
||||
- text: "<|startoftext|>החוק השני של מועדון קרב הוא"
|
||||
- text: "<|startoftext|>ראש הממשלה בן גוריון"
|
||||
- text: "<|startoftext|>למידת מכונה (סרט)"
|
||||
- text: "<|startoftext|>מנשה פומפרניקל"
|
||||
- text: "<|startoftext|>אי שוויון "
|
||||
|
||||
license: mit
|
||||
---
|
||||
|
||||
|
||||
# hewiki-articles-distilGPT2py-il
|
||||
|
||||
## A tiny GPT2 model for generating Hebrew text
|
||||
|
||||
A distilGPT2 sized model. <br>
|
||||
Training data was hewiki-20200701-pages-articles-multistream.xml.bz2 from https://dumps.wikimedia.org/hewiki/20200701/ <br>
|
||||
XML has been converted to plain text using Wikipedia Extractor http://medialab.di.unipi.it/wiki/Wikipedia_Extractor <br>
|
||||
I then added <|startoftext|> and <|endoftext|> markers and deleted empty lines. <br>
|
||||
|
||||
#### How to use
|
||||
|
||||
```python
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
from transformers import GPT2Tokenizer, GPT2LMHeadModel
|
||||
|
||||
tokenizer = GPT2Tokenizer.from_pretrained("Norod78/hewiki-articles-distilGPT2py-il")
|
||||
model = GPT2LMHeadModel.from_pretrained("Norod78/hewiki-articles-distilGPT2py-il").eval()
|
||||
|
||||
bos_token = tokenizer.bos_token #Beginning of sentace
|
||||
eos_token = tokenizer.eos_token #End of sentence
|
||||
|
||||
def generate_word(model, tokens_tensor, temperature=1.0):
|
||||
"""
|
||||
Sample a word given a tensor of tokens of previous words from a model. Given
|
||||
the words we have, sample a plausible word. Temperature is used for
|
||||
controlling randomness. If using temperature==0 we simply use a greedy arg max.
|
||||
Else, we sample from a multinomial distribution using a lower inverse
|
||||
temperature to allow for more randomness to escape repetitions.
|
||||
"""
|
||||
with torch.no_grad():
|
||||
outputs = model(tokens_tensor)
|
||||
predictions = outputs[0]
|
||||
if temperature>0:
|
||||
# Make the distribution more or less skewed based on the temperature
|
||||
predictions = outputs[0]/temperature
|
||||
# Sample from the distribution
|
||||
softmax = nn.Softmax(dim=0)
|
||||
predicted_index = torch.multinomial(softmax(predictions[0,-1,:]),1).item()
|
||||
# Simply take the arg-max of the distribution
|
||||
else:
|
||||
predicted_index = torch.argmax(predictions[0, -1, :]).item()
|
||||
# Decode the encoding to the corresponding word
|
||||
predicted_text = tokenizer.decode([predicted_index])
|
||||
return predicted_text
|
||||
|
||||
def generate_sentence(model, tokenizer, initial_text, temperature=1.0):
|
||||
""" Generate a sentence given some initial text using a model and a tokenizer.
|
||||
Returns the new sentence. """
|
||||
|
||||
# Encode a text inputs
|
||||
text = ""
|
||||
sentence = text
|
||||
|
||||
# We avoid an infinite loop by setting a maximum range
|
||||
for i in range(0,84):
|
||||
indexed_tokens = tokenizer.encode(initial_text + text)
|
||||
|
||||
# Convert indexed tokens in a PyTorch tensor
|
||||
tokens_tensor = torch.tensor([indexed_tokens])
|
||||
|
||||
new_word = generate_word(model, tokens_tensor, temperature=temperature)
|
||||
|
||||
# Here the temperature is slowly decreased with each generated word,
|
||||
# this ensures that the sentence (ending) makes more sense.
|
||||
# We don't decrease to a temperature of 0.0 to leave some randomness in.
|
||||
if temperature<(1-0.008):
|
||||
temperature += 0.008
|
||||
else:
|
||||
temperature = 0.996
|
||||
|
||||
text = text+new_word
|
||||
|
||||
# Stop generating new words when we have reached the end of the line or the poem
|
||||
if eos_token in new_word:
|
||||
# returns new sentence and whether poem is done
|
||||
return (text.replace(eos_token,"").strip(), True)
|
||||
elif '/' in new_word:
|
||||
return (text.strip(), False)
|
||||
elif bos_token in new_word:
|
||||
return (text.replace(bos_token,"").strip(), False)
|
||||
|
||||
return (text, True)
|
||||
|
||||
for output_num in range(1,5):
|
||||
init_text = "בוקר טוב"
|
||||
text = bos_token + init_text
|
||||
for i in range(0,84):
|
||||
sentence = generate_sentence(model, tokenizer, text, temperature=0.9)
|
||||
text = init_text + sentence[0]
|
||||
print(text)
|
||||
if (sentence[1] == True):
|
||||
break
|
||||
```
|
||||
@@ -1,48 +0,0 @@
|
||||
---
|
||||
language:
|
||||
- ach
|
||||
- en
|
||||
tags:
|
||||
- translation
|
||||
license: cc-by-4.0
|
||||
datasets:
|
||||
- JW300
|
||||
metrics:
|
||||
- bleu
|
||||
---
|
||||
|
||||
# HEL-ACH-EN
|
||||
|
||||
## Model description
|
||||
|
||||
MT model translating Acholi to English initialized with weights from [opus-mt-luo-en](https://huggingface.co/Helsinki-NLP/opus-mt-luo-en) on HuggingFace.
|
||||
|
||||
## Intended uses & limitations
|
||||
Machine Translation experiments. Do not use for sensitive tasks.
|
||||
#### How to use
|
||||
|
||||
```python
|
||||
# You can include sample code which will be formatted
|
||||
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Ogayo/Hel-ach-en")
|
||||
|
||||
model = AutoModelForSeq2SeqLM.from_pretrained("Ogayo/Hel-ach-en")
|
||||
|
||||
```
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
Trained on Jehovah Witnesses data so contains theirs and Christian views.
|
||||
|
||||
## Training data
|
||||
Trained on OPUS JW300 data.
|
||||
Initialized with weights from [opus-mt-luo-en](https://huggingface.co/Helsinki-NLP/opus-mt-luo-en?text=Bed+gi+nyasi+mar+chieng%27+nyuol+mopong%27+gi+mor%21#model_card)
|
||||
|
||||
## Training procedure
|
||||
|
||||
Remove duplicates and rows with no alphabetic characters. Used GPU
|
||||
## Eval results
|
||||
testset | BLEU
|
||||
--- | ---
|
||||
JW300.luo.en| 46.1
|
||||
@@ -1,63 +0,0 @@
|
||||
---
|
||||
language: "en"
|
||||
---
|
||||
|
||||
# BART-Squad2
|
||||
|
||||
## Model description
|
||||
|
||||
BART for extractive (span-based) question answering, trained on Squad 2.0.
|
||||
|
||||
F1 score of 87.4.
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
Unfortunately, the Huggingface auto-inference API won't run this model, so if you're attempting to try it through the input box above and it complains, don't be discouraged!
|
||||
|
||||
#### How to use
|
||||
|
||||
Here's a quick way to get question answering running locally:
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModelForQuestionAnswering
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained("Primer/bart-squad2")
|
||||
model = AutoModelForQuestionAnswering.from_pretrained("Primer/bart-squad2")
|
||||
model.to('cuda'); model.eval()
|
||||
|
||||
def answer(question, text):
|
||||
seq = '<s>' + question + ' </s> </s> ' + text + ' </s>'
|
||||
tokens = tokenizer.encode_plus(seq, return_tensors='pt', padding='max_length', max_length=1024)
|
||||
input_ids = tokens['input_ids'].to('cuda')
|
||||
attention_mask = tokens['attention_mask'].to('cuda')
|
||||
start, end, _ = model(input_ids, attention_mask=attention_mask)
|
||||
start_idx = int(start.argmax().int())
|
||||
end_idx = int(end.argmax().int())
|
||||
print(tokenizer.decode(input_ids[0, start_idx:end_idx]).strip())
|
||||
# ^^ it will be an empty string if the model decided "unanswerable"
|
||||
|
||||
>>> question = "Where does Tom live?"
|
||||
>>> context = "Tom is an engineer in San Francisco."
|
||||
>>> answer(question, context)
|
||||
San Francisco
|
||||
```
|
||||
|
||||
(Just drop the `.to('cuda')` stuff if running on CPU).
|
||||
|
||||
#### Limitations and bias
|
||||
|
||||
Unknown, no further evaluation has been performed. In a technical sense one big limitation is that it's 1.6G 😬
|
||||
|
||||
## Training procedure
|
||||
|
||||
`run_squad.py` with:
|
||||
|
||||
|param|value|
|
||||
|---|---|
|
||||
|batch size|8|
|
||||
|max_seq_length|1024|
|
||||
|learning rate|1e-5|
|
||||
|epochs|2|
|
||||
|
||||
Modified to freeze shared parameters and encoder embeddings.
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
## 🔥 Model cards now live inside each huggingface.co model repo 🔥
|
||||
|
||||
|
||||
For consistency, ease of use and scalability, `README.md` model cards now live directly inside each model repo on the HuggingFace model hub.
|
||||
|
||||
### How to update a model card
|
||||
|
||||
You can directly update a model card inside any model repo you have **write access** to, i.e.:
|
||||
- a model under your username namespace
|
||||
- a model under any organization you are a part of.
|
||||
|
||||
You can either:
|
||||
- update it, commit and push using your usual git workflow (command line, GUI, etc.)
|
||||
- or edit it directly from the website's UI.
|
||||
|
||||
**What if you want to create or update a model card for a model you don't have write access to?**
|
||||
|
||||
In that case, given that we don't have a Pull request system yet on huggingface.co (🤯),
|
||||
you can open an issue here, post the card's content, and tag the model author(s) and/or the Hugging Face team.
|
||||
|
||||
We might implement a more seamless process at some point, so your early feedback is precious!
|
||||
Please let us know of any suggestion.
|
||||
|
||||
### What happened to the model cards here?
|
||||
|
||||
We migrated every model card from the repo to its corresponding huggingface.co model repo. Individual commits were preserved, and they link back to the original commit on GitHub.
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user