Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fd1f2115cf | ||
|
|
c65098832a | ||
|
|
db3a6ab640 |
-12
@@ -1,12 +0,0 @@
|
||||
[run]
|
||||
source=transformers
|
||||
omit =
|
||||
# skip convertion scripts from testing for now
|
||||
*/convert_*
|
||||
*/__main__.py
|
||||
[report]
|
||||
exclude_lines =
|
||||
pragma: no cover
|
||||
raise
|
||||
except
|
||||
register_parameter
|
||||
@@ -0,0 +1,200 @@
|
||||
- ALBERT:
|
||||
paper: ALBERT: A Lite BERT for Self-supervised Learning of Language Representations, by Zhenzhong Lan, Mingda Chen, Sebastian Goodman, Kevin Gimpel, Piyush Sharma, Radu Soricut
|
||||
tags:
|
||||
- encoder
|
||||
- memory-efficient
|
||||
- weight-sharing
|
||||
|
||||
- BART:
|
||||
paper: BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension by Mike Lewis, Yinhan Liu, Naman Goyal, Marjan Ghazvininejad, Abdelrahman Mohamed, Omer Levy, Ves Stoyanov and Luke Zettlemoyer.
|
||||
tags:
|
||||
- encoder-decoder
|
||||
- denoising
|
||||
|
||||
- BERT:
|
||||
paper: BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding by Jacob Devlin, Ming-Wei Chang, Kenton Lee and Kristina Toutanova.
|
||||
tags:
|
||||
- encoder
|
||||
- bert
|
||||
|
||||
- BERT For Sequence Generation:
|
||||
paper: Leveraging Pre-trained Checkpoints for Sequence Generation Tasks by Sascha Rothe, Shashi Narayan, Aliaksei Severyn.
|
||||
tags:
|
||||
- encoder-decoder
|
||||
- bert
|
||||
|
||||
- Blenderbot:
|
||||
paper: Recipes for building an open-domain chatbot by Stephen Roller, Emily Dinan, Naman Goyal, Da Ju, Mary Williamson, Yinhan Liu, Jing Xu, Myle Ott, Kurt Shuster, Eric M. Smith, Y-Lan Boureau, Jason Weston.
|
||||
tags:
|
||||
- encoder-decoder
|
||||
- chatbot
|
||||
- conditional-generation
|
||||
|
||||
- CamemBERT:
|
||||
paper: CamemBERT: a Tasty French Language Model by Louis Martin*, Benjamin Muller*, Pedro Javier Ortiz Suárez*, Yoann Dupont, Laurent Romary, Éric Villemonte de la Clergerie, Djamé Seddah and Benoît Sagot.
|
||||
tags:
|
||||
- encoder
|
||||
- french
|
||||
- bert
|
||||
|
||||
- CTRL:
|
||||
paper: CTRL: A Conditional Transformer Language Model for Controllable Generation by Nitish Shirish Keskar*, Bryan McCann*, Lav R. Varshney, Caiming Xiong and Richard Socher.
|
||||
tags:
|
||||
- decoder
|
||||
- conditional-generation
|
||||
|
||||
- DeBERTa:
|
||||
paper: DeBERTa: Decoding-enhanced BERT with Disentangled Attention by Pengcheng He, Xiaodong Liu, Jianfeng Gao, Weizhu Chen.
|
||||
tags:
|
||||
|
||||
- DialoGPT:
|
||||
paper: DialoGPT: Large-Scale Generative Pre-training for Conversational Response Generation by Yizhe Zhang, Siqi Sun, Michel Galley, Yen-Chun Chen, Chris Brockett, Xiang Gao, Jianfeng Gao, Jingjing Liu, Bill Dolan.
|
||||
tags:
|
||||
- decoder
|
||||
- chatbot
|
||||
- conditional-generation
|
||||
|
||||
- DistilBERT:
|
||||
paper: DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter by Victor Sanh, Lysandre Debut and Thomas Wolf. The same method has been applied to compress GPT2 into DistilGPT2, RoBERTa into DistilRoBERTa, Multilingual BERT into DistilmBERT and a German version of DistilBERT.
|
||||
tags:
|
||||
- encoder
|
||||
- efficient
|
||||
- distillation
|
||||
|
||||
- DPR:
|
||||
paper: Dense Passage Retrieval for Open-Domain Question Answering by Vladimir Karpukhin, Barlas Oğuz, Sewon Min, Patrick Lewis, Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih.
|
||||
tags:
|
||||
- encoder
|
||||
- retrieval
|
||||
|
||||
- ELECTRA:
|
||||
paper: ELECTRA: Pre-training text encoders as discriminators rather than generators by Kevin Clark, Minh-Thang Luong, Quoc V. Le, Christopher D. Manning.
|
||||
tags:
|
||||
- encoder
|
||||
- efficient-training
|
||||
|
||||
- FlauBERT:
|
||||
paper: FlauBERT: Unsupervised Language Model Pre-training for French by Hang Le, Loïc Vial, Jibril Frej, Vincent Segonne, Maximin Coavoux, Benjamin Lecouteux, Alexandre Allauzen, Benoît Crabbé, Laurent Besacier, Didier Schwab.
|
||||
tags:
|
||||
- encoder
|
||||
- french
|
||||
|
||||
- Funnel Transformer:
|
||||
paper: Funnel-Transformer: Filtering out Sequential Redundancy for Efficient Language Processing by Zihang Dai, Guokun Lai, Yiming Yang, Quoc V. Le.
|
||||
tags:
|
||||
- efficient
|
||||
- bottleneck
|
||||
|
||||
- GPT:
|
||||
paper: Improving Language Understanding by Generative Pre-Training by Alec Radford, Karthik Narasimhan, Tim Salimans and Ilya Sutskever.
|
||||
tags:
|
||||
- decoder
|
||||
- unconditional-generation
|
||||
|
||||
- GPT-2:
|
||||
paper: Language Models are Unsupervised Multitask Learners by Alec Radford*, Jeffrey Wu*, Rewon Child, David Luan, Dario Amodei** and Ilya Sutskever**.
|
||||
tags:
|
||||
- decoder
|
||||
- unconditional-generation
|
||||
- over-1-billion-parameters
|
||||
|
||||
- LayoutLM:
|
||||
paper: LayoutLM: Pre-training of Text and Layout for Document Image Understanding by Yiheng Xu, Minghao Li, Lei Cui, Shaohan Huang, Furu Wei, Ming Zhou.
|
||||
tags:
|
||||
|
||||
- Longformer:
|
||||
paper: Longformer: The Long-Document Transformer by Iz Beltagy, Matthew E. Peters, Arman Cohan.
|
||||
tags:
|
||||
- encoder
|
||||
- efficient-attention
|
||||
- data-independant-pattern
|
||||
- long-inputs
|
||||
|
||||
- LXMERT:
|
||||
paper: LXMERT: Learning Cross-Modality Encoder Representations from Transformers for Open-Domain Question Answering by Hao Tan and Mohit Bansal.
|
||||
tags:
|
||||
- encoder
|
||||
- multi-modal
|
||||
- image
|
||||
- text
|
||||
|
||||
- MarianMT:
|
||||
paper: Machine translation models trained using OPUS data by Jörg Tiedemann. The Marian Framework is being developed by the Microsoft Translator Team.
|
||||
tags:
|
||||
- encoder-decoder``
|
||||
- translation
|
||||
|
||||
- MBart:
|
||||
paper: Multilingual Denoising Pre-training for Neural Machine Translation by Yinhan Liu, Jiatao Gu, Naman Goyal, Xian Li, Sergey Edunov, Marjan Ghazvininejad, Mike Lewis, Luke Zettlemoyer.
|
||||
tags:
|
||||
- encoder-decoder
|
||||
- multi-lingual
|
||||
- denoising
|
||||
|
||||
- Pegasus:
|
||||
paper: PEGASUS: Pre-training with Extracted Gap-sentences for Abstractive Summarization> by Jingqing Zhang, Yao Zhao, Mohammad Saleh and Peter J. Liu.
|
||||
tags:
|
||||
- efficient-pretraining
|
||||
|
||||
- ProphetNet:
|
||||
paper: ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training by Yu Yan, Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang and Ming Zhou.
|
||||
tags:
|
||||
- encoder-decoder
|
||||
- future-prediction
|
||||
|
||||
- Reformer:
|
||||
paper: Reformer: The Efficient Transformer by Nikita Kitaev, Łukasz Kaiser, Anselm Levskaya.
|
||||
tags:
|
||||
- encoder
|
||||
- efficient-attention
|
||||
- data-dependant-pattern
|
||||
- long-inputs
|
||||
|
||||
- RoBERTa:
|
||||
paper: a Robustly Optimized BERT Pretraining Approach by Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, Veselin Stoyanov. ultilingual BERT into DistilmBERT and a German version of DistilBERT.
|
||||
tags:
|
||||
- encoder
|
||||
- large-dataset-training
|
||||
|
||||
- SqueezeBert:
|
||||
paper SqueezeBERT: What can computer vision teach NLP about efficient neural networks? by Forrest N. Iandola, Albert E. Shaw, Ravi Krishna, and Kurt W. Keutzer.
|
||||
tags:
|
||||
|
||||
- T5:
|
||||
paper: Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer by Colin Raffel and Noam Shazeer and Adam Roberts and Katherine Lee and Sharan Narang and Michael Matena and Yanqi Zhou and Wei Li and Peter J. Liu.
|
||||
tags:
|
||||
- encoder-decoder
|
||||
- large-dataset-training
|
||||
- over-1-billion-parameters
|
||||
|
||||
- Transformer-XL:
|
||||
paper: Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context by Zihang Dai*, Zhilin Yang*, Yiming Yang, Jaime Carbonell, Quoc V. Le, Ruslan Salakhutdinov.
|
||||
tags:
|
||||
- decoder
|
||||
- long-inputs
|
||||
|
||||
- XLM:
|
||||
paper: Cross-lingual Language Model Pretraining by Guillaume Lample and Alexis Conneau.
|
||||
tags:
|
||||
- encoder
|
||||
- multilingual
|
||||
|
||||
- XLM-ProphetNet:
|
||||
paper: ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training by Yu Yan, Weizhen Qi, Yeyun Gong, Dayiheng Liu, Nan Duan, Jiusheng Chen, Ruofei Zhang and Ming Zhou.
|
||||
tags:
|
||||
- encoder-decoder
|
||||
- future-prediction
|
||||
- mutlilingual
|
||||
|
||||
- XLM-RoBERTa:
|
||||
paper: Unsupervised Cross-lingual Representation Learning at Scale by Alexis Conneau*, Kartikay Khandelwal*, Naman Goyal, Vishrav Chaudhary, Guillaume Wenzek, Francisco Guzmán, Edouard Grave, Myle Ott, Luke Zettlemoyer and Veselin Stoyanov.
|
||||
tags:
|
||||
- encoder
|
||||
- mutlilingual
|
||||
|
||||
- XLNet:
|
||||
paper: XLNet: Generalized Autoregressive Pretraining for Language Understanding by Zhilin Yang*, Zihang Dai*, Yiming Yang, Jaime Carbonell, Ruslan Salakhutdinov, Quoc V. Le.
|
||||
tags:
|
||||
- decoder
|
||||
- bidirectional
|
||||
- auto-regressive
|
||||
@@ -1,91 +0,0 @@
|
||||
---
|
||||
|
||||
- step:
|
||||
name: Execute python examples/text-classification/run_glue.py
|
||||
image: pytorch/pytorch:nightly-devel-cuda10.0-cudnn7
|
||||
command:
|
||||
- python /valohai/repository/utils/download_glue_data.py --data_dir=/glue_data
|
||||
- pip install -e .
|
||||
- pip install -r examples/requirements.txt
|
||||
- python examples/text-classification/run_glue.py --do_train --data_dir=/glue_data/{parameter-value:task_name} {parameters}
|
||||
parameters:
|
||||
- name: model_type
|
||||
pass-as: --model_type={v}
|
||||
type: string
|
||||
default: bert
|
||||
- name: model_name_or_path
|
||||
pass-as: --model_name_or_path={v}
|
||||
type: string
|
||||
default: bert-base-uncased
|
||||
- name: task_name
|
||||
pass-as: --task_name={v}
|
||||
type: string
|
||||
default: MRPC
|
||||
- name: max_seq_length
|
||||
pass-as: --max_seq_length={v}
|
||||
description: The maximum total input sequence length after tokenization. Sequences longer than this will be truncated, sequences shorter will be padded.
|
||||
type: integer
|
||||
default: 128
|
||||
- name: per_gpu_train_batch_size
|
||||
pass-as: --per_gpu_train_batch_size={v}
|
||||
description: Batch size per GPU/CPU for training.
|
||||
type: integer
|
||||
default: 8
|
||||
- name: per_gpu_eval_batch_size
|
||||
pass-as: --per_gpu_eval_batch_size={v}
|
||||
description: Batch size per GPU/CPU for evaluation.
|
||||
type: integer
|
||||
default: 8
|
||||
- name: gradient_accumulation_steps
|
||||
pass-as: --gradient_accumulation_steps={v}
|
||||
description: Number of updates steps to accumulate before performing a backward/update pass.
|
||||
type: integer
|
||||
default: 1
|
||||
- name: learning_rate
|
||||
pass-as: --learning_rate={v}
|
||||
description: The initial learning rate for Adam.
|
||||
type: float
|
||||
default: 0.00005
|
||||
- name: adam_epsilon
|
||||
pass-as: --adam_epsilon={v}
|
||||
description: Epsilon for Adam optimizer.
|
||||
type: float
|
||||
default: 0.00000001
|
||||
- name: max_grad_norm
|
||||
pass-as: --max_grad_norm={v}
|
||||
description: Max gradient norm.
|
||||
type: float
|
||||
default: 1.0
|
||||
- name: num_train_epochs
|
||||
pass-as: --num_train_epochs={v}
|
||||
description: Total number of training epochs to perform.
|
||||
type: integer
|
||||
default: 3
|
||||
- name: max_steps
|
||||
pass-as: --max_steps={v}
|
||||
description: If > 0, set total number of training steps to perform. Override num_train_epochs.
|
||||
type: integer
|
||||
default: -1
|
||||
- name: warmup_steps
|
||||
pass-as: --warmup_steps={v}
|
||||
description: Linear warmup over warmup_steps.
|
||||
type: integer
|
||||
default: -1
|
||||
- name: logging_steps
|
||||
pass-as: --logging_steps={v}
|
||||
description: Log every X updates steps.
|
||||
type: integer
|
||||
default: 25
|
||||
- name: save_steps
|
||||
pass-as: --save_steps={v}
|
||||
description: Save checkpoint every X updates steps.
|
||||
type: integer
|
||||
default: -1
|
||||
- name: output_dir
|
||||
pass-as: --output_dir={v}
|
||||
type: string
|
||||
default: /valohai/outputs
|
||||
- name: evaluation_strategy
|
||||
description: The evaluation strategy to use.
|
||||
type: string
|
||||
default: steps
|
||||
Reference in New Issue
Block a user