Compare commits
16
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b9b777749b | ||
|
|
79eb391586 | ||
|
|
7087d9b1c0 | ||
|
|
efc4a21ffa | ||
|
|
5148f43309 | ||
|
|
38f6739cd6 | ||
|
|
00602f7840 | ||
|
|
3c682ea15c | ||
|
|
59b5953d89 | ||
|
|
6e07c1f446 | ||
|
|
43fdafef89 | ||
|
|
627e813734 | ||
|
|
9865e1fe52 | ||
|
|
d39da5a2ab | ||
|
|
5e323017a4 | ||
|
|
4acfd1a8dc |
No files matched your search
@@ -187,7 +187,7 @@ def train(args, train_dataset, model, tokenizer):
|
||||
"end_positions": batch[4],
|
||||
}
|
||||
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart"]:
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart", "longformer"]:
|
||||
del inputs["token_type_ids"]
|
||||
|
||||
if args.model_type in ["xlnet", "xlm"]:
|
||||
@@ -300,7 +300,7 @@ def evaluate(args, model, tokenizer, prefix=""):
|
||||
"token_type_ids": batch[2],
|
||||
}
|
||||
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart"]:
|
||||
if args.model_type in ["xlm", "roberta", "distilbert", "camembert", "bart", "longformer"]:
|
||||
del inputs["token_type_ids"]
|
||||
|
||||
feature_indices = batch[3]
|
||||
|
||||
@@ -16,7 +16,6 @@ from transformers import (
|
||||
)
|
||||
from transformers.trainer_utils import EvaluationStrategy
|
||||
from utils import (
|
||||
LegacySeq2SeqDataset,
|
||||
Seq2SeqDataCollator,
|
||||
Seq2SeqDataset,
|
||||
assert_all_frozen,
|
||||
@@ -138,6 +137,10 @@ class DataTrainingArguments:
|
||||
src_lang: Optional[str] = field(default=None, metadata={"help": "Source language id for translation."})
|
||||
tgt_lang: Optional[str] = field(default=None, metadata={"help": "Target language id for translation."})
|
||||
eval_beams: Optional[int] = field(default=None, metadata={"help": "# num_beams to use for evaluation."})
|
||||
ignore_pad_token_for_loss: bool = field(
|
||||
default=True,
|
||||
metadata={"help": "If only pad tokens should be ignored. This assumes that `config.pad_token_id` is defined."},
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
@@ -223,7 +226,7 @@ def main():
|
||||
freeze_params(model.get_encoder())
|
||||
assert_all_frozen(model.get_encoder())
|
||||
|
||||
dataset_class = Seq2SeqDataset if hasattr(tokenizer, "prepare_seq2seq_batch") else LegacySeq2SeqDataset
|
||||
dataset_class = Seq2SeqDataset
|
||||
|
||||
# Get datasets
|
||||
train_dataset = (
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
import logging
|
||||
import copy
|
||||
from typing import Any, Dict, Optional, Tuple, Union
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
from torch.utils.data import DistributedSampler, RandomSampler
|
||||
|
||||
from transformers import Trainer
|
||||
from transformers import PreTrainedModel, Trainer, logging
|
||||
from transformers.configuration_fsmt import FSMTConfig
|
||||
from transformers.file_utils import is_torch_tpu_available
|
||||
from transformers.optimization import (
|
||||
@@ -27,7 +27,7 @@ except ImportError:
|
||||
from utils import label_smoothed_nll_loss
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
arg_to_scheduler = {
|
||||
"linear": get_linear_schedule_with_warmup,
|
||||
@@ -41,13 +41,25 @@ arg_to_scheduler_choices = sorted(arg_to_scheduler.keys())
|
||||
|
||||
|
||||
class Seq2SeqTrainer(Trainer):
|
||||
def __init__(self, config, data_args, *args, **kwargs):
|
||||
def __init__(self, config=None, data_args=None, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
self.config = config
|
||||
|
||||
if config is None:
|
||||
assert isinstance(
|
||||
self.model, PreTrainedModel
|
||||
), f"If no `config` is passed the model to be trained has to be of type `PreTrainedModel`, but is {self.model.__class__}"
|
||||
self.config = self._actual_model(self.model).config
|
||||
else:
|
||||
self.config = config
|
||||
|
||||
self.data_args = data_args
|
||||
self.max_gen_length = data_args.val_max_target_length
|
||||
self.vocab_size = self.config.tgt_vocab_size if isinstance(self.config, FSMTConfig) else self.config.vocab_size
|
||||
|
||||
if self.args.label_smoothing != 0 or (self.data_args is not None and self.data_args.ignore_pad_token_for_loss):
|
||||
assert (
|
||||
self.config.pad_token_id is not None
|
||||
), "Make sure that `config.pad_token_id` is correcly defined when ignoring `pad_token` for loss calculation or doing label smoothing."
|
||||
|
||||
def create_optimizer_and_scheduler(self, num_training_steps: int):
|
||||
"""
|
||||
Setup the optimizer and the learning rate scheduler.
|
||||
@@ -114,23 +126,31 @@ class Seq2SeqTrainer(Trainer):
|
||||
else DistributedSampler(self.train_dataset)
|
||||
)
|
||||
|
||||
def compute_loss(self, model, inputs):
|
||||
labels = inputs.pop("labels")
|
||||
outputs = model(**inputs, use_cache=False)
|
||||
logits = outputs[0]
|
||||
return self._compute_loss(logits, labels)
|
||||
|
||||
def _compute_loss(self, logits, labels):
|
||||
def _compute_loss(self, model, inputs):
|
||||
inputs = copy.deepcopy(inputs)
|
||||
if self.args.label_smoothing == 0:
|
||||
# Same behavior as modeling_bart.py
|
||||
loss_fct = torch.nn.CrossEntropyLoss(ignore_index=self.config.pad_token_id)
|
||||
assert logits.shape[-1] == self.vocab_size
|
||||
loss = loss_fct(logits.view(-1, logits.shape[-1]), labels.view(-1))
|
||||
if self.data_args is not None and self.data_args.ignore_pad_token_for_loss:
|
||||
# force training to ignore pad token
|
||||
labels = inputs.pop("labels")
|
||||
logits = model(**inputs, use_cache=False)[0]
|
||||
|
||||
loss_fct = torch.nn.CrossEntropyLoss(ignore_index=self.config.pad_token_id)
|
||||
loss = loss_fct(logits.view(-1, logits.shape[-1]), labels.view(-1))
|
||||
else:
|
||||
# compute usual loss via models
|
||||
loss, logits = model(**inputs, use_cache=False)[:2]
|
||||
else:
|
||||
# compute label smoothed loss
|
||||
labels = inputs.pop("labels")
|
||||
logits = model(**inputs, use_cache=False)[0]
|
||||
lprobs = torch.nn.functional.log_softmax(logits, dim=-1)
|
||||
loss, nll_loss = label_smoothed_nll_loss(
|
||||
loss, _ = label_smoothed_nll_loss(
|
||||
lprobs, labels, self.args.label_smoothing, ignore_index=self.config.pad_token_id
|
||||
)
|
||||
return loss, logits
|
||||
|
||||
def compute_loss(self, model, inputs):
|
||||
loss, _ = self._compute_loss(model, inputs)
|
||||
return loss
|
||||
|
||||
def prediction_step(
|
||||
@@ -158,31 +178,37 @@ class Seq2SeqTrainer(Trainer):
|
||||
"""
|
||||
inputs = self._prepare_inputs(inputs)
|
||||
|
||||
if self.args.predict_with_generate and not self.args.prediction_loss_only:
|
||||
gen_kwargs = {
|
||||
"max_length": self.data_args.val_max_target_length
|
||||
if self.data_args is not None
|
||||
else self.config.max_length,
|
||||
"num_beams": self.data_args.eval_beams if self.data_args is not None else self.config.num_beams,
|
||||
}
|
||||
generated_tokens = model.generate(
|
||||
inputs["input_ids"],
|
||||
attention_mask=inputs["attention_mask"],
|
||||
**gen_kwargs,
|
||||
)
|
||||
# in case the batch is shorter than max length, the output should be padded
|
||||
if self.config.pad_token_id is not None:
|
||||
generated_tokens = self._pad_tensors_to_max_len(generated_tokens, gen_kwargs["max_length"])
|
||||
|
||||
# compute loss on predict data
|
||||
with torch.no_grad():
|
||||
if self.args.predict_with_generate and not self.args.prediction_loss_only:
|
||||
generated_tokens = model.generate(
|
||||
inputs["input_ids"],
|
||||
attention_mask=inputs["attention_mask"],
|
||||
use_cache=True,
|
||||
num_beams=self.data_args.eval_beams,
|
||||
max_length=self.max_gen_length,
|
||||
)
|
||||
# in case the batch is shorter than max length, the output should be padded
|
||||
generated_tokens = self._pad_tensors_to_max_len(generated_tokens, self.max_gen_length)
|
||||
loss, logits = self._compute_loss(model, inputs)
|
||||
|
||||
labels_out = inputs.get("labels")
|
||||
# Call forward again to get loss # TODO: avoidable?
|
||||
outputs = model(**inputs, use_cache=False)
|
||||
loss = self._compute_loss(outputs[1], labels_out)
|
||||
loss = loss.mean().detach()
|
||||
if self.args.prediction_loss_only:
|
||||
return (loss, None, None)
|
||||
loss = loss.mean().detach()
|
||||
if self.args.prediction_loss_only:
|
||||
return (loss, None, None)
|
||||
|
||||
logits = generated_tokens if self.args.predict_with_generate else outputs[1]
|
||||
logits = generated_tokens if self.args.predict_with_generate else logits
|
||||
|
||||
labels_out = labels_out.detach()
|
||||
labels = self._pad_tensors_to_max_len(labels_out, self.max_gen_length)
|
||||
return (loss, logits.detach(), labels)
|
||||
labels = inputs["labels"]
|
||||
if self.config.pad_token_id is not None:
|
||||
labels = self._pad_tensors_to_max_len(labels, self.config.max_length)
|
||||
|
||||
return (loss, logits, labels)
|
||||
|
||||
def _pad_tensors_to_max_len(self, tensor, max_length):
|
||||
padded_tensor = self.config.pad_token_id * torch.ones(
|
||||
|
||||
@@ -5,12 +5,14 @@ from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from transformers import is_torch_available
|
||||
from transformers import BertTokenizer, EncoderDecoderModel, is_torch_available
|
||||
from transformers.file_utils import is_datasets_available
|
||||
from transformers.testing_utils import TestCasePlus, slow
|
||||
from transformers.trainer_callback import TrainerState
|
||||
from transformers.trainer_utils import set_seed
|
||||
|
||||
from .finetune_trainer import main
|
||||
from .finetune_trainer import Seq2SeqTrainingArguments, main
|
||||
from .seq2seq_trainer import Seq2SeqTrainer
|
||||
from .test_seq2seq_examples import MBART_TINY
|
||||
from .utils import execute_async_std
|
||||
|
||||
@@ -50,6 +52,117 @@ class TestFinetuneTrainer(TestCasePlus):
|
||||
assert "test_generations.txt" in contents
|
||||
assert "test_results.json" in contents
|
||||
|
||||
@slow
|
||||
def test_finetune_bert2bert(self):
|
||||
if not is_datasets_available():
|
||||
return
|
||||
|
||||
import datasets
|
||||
|
||||
bert2bert = EncoderDecoderModel.from_encoder_decoder_pretrained("prajjwal1/bert-tiny", "prajjwal1/bert-tiny")
|
||||
tokenizer = BertTokenizer.from_pretrained("bert-base-uncased")
|
||||
|
||||
bert2bert.config.vocab_size = bert2bert.config.encoder.vocab_size
|
||||
bert2bert.config.decoder_start_token_id = tokenizer.cls_token_id
|
||||
|
||||
train_dataset = datasets.load_dataset("cnn_dailymail", "3.0.0", split="train[:1%]")
|
||||
val_dataset = datasets.load_dataset("cnn_dailymail", "3.0.0", split="validation[:1%]")
|
||||
|
||||
train_dataset = train_dataset.select(range(32))
|
||||
val_dataset = val_dataset.select(range(16))
|
||||
|
||||
rouge = datasets.load_metric("rouge")
|
||||
|
||||
batch_size = 4
|
||||
|
||||
def _map_to_encoder_decoder_inputs(batch):
|
||||
# Tokenizer will automatically set [BOS] <text> [EOS]
|
||||
inputs = tokenizer(batch["article"], padding="max_length", truncation=True, max_length=512)
|
||||
outputs = tokenizer(batch["highlights"], padding="max_length", truncation=True, max_length=128)
|
||||
batch["input_ids"] = inputs.input_ids
|
||||
batch["attention_mask"] = inputs.attention_mask
|
||||
|
||||
batch["decoder_input_ids"] = outputs.input_ids
|
||||
batch["labels"] = outputs.input_ids.copy()
|
||||
batch["labels"] = [
|
||||
[-100 if token == tokenizer.pad_token_id else token for token in labels] for labels in batch["labels"]
|
||||
]
|
||||
batch["decoder_attention_mask"] = outputs.attention_mask
|
||||
|
||||
assert all([len(x) == 512 for x in inputs.input_ids])
|
||||
assert all([len(x) == 128 for x in outputs.input_ids])
|
||||
|
||||
return batch
|
||||
|
||||
def _compute_metrics(pred):
|
||||
labels_ids = pred.label_ids
|
||||
pred_ids = pred.predictions
|
||||
|
||||
# all unnecessary tokens are removed
|
||||
pred_str = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)
|
||||
label_str = tokenizer.batch_decode(labels_ids, skip_special_tokens=True)
|
||||
|
||||
rouge_output = rouge.compute(predictions=pred_str, references=label_str, rouge_types=["rouge2"])[
|
||||
"rouge2"
|
||||
].mid
|
||||
|
||||
return {
|
||||
"rouge2_precision": round(rouge_output.precision, 4),
|
||||
"rouge2_recall": round(rouge_output.recall, 4),
|
||||
"rouge2_fmeasure": round(rouge_output.fmeasure, 4),
|
||||
}
|
||||
|
||||
# map train dataset
|
||||
train_dataset = train_dataset.map(
|
||||
_map_to_encoder_decoder_inputs,
|
||||
batched=True,
|
||||
batch_size=batch_size,
|
||||
remove_columns=["article", "highlights"],
|
||||
)
|
||||
train_dataset.set_format(
|
||||
type="torch",
|
||||
columns=["input_ids", "attention_mask", "decoder_input_ids", "decoder_attention_mask", "labels"],
|
||||
)
|
||||
|
||||
# same for validation dataset
|
||||
val_dataset = val_dataset.map(
|
||||
_map_to_encoder_decoder_inputs,
|
||||
batched=True,
|
||||
batch_size=batch_size,
|
||||
remove_columns=["article", "highlights"],
|
||||
)
|
||||
val_dataset.set_format(
|
||||
type="torch",
|
||||
columns=["input_ids", "attention_mask", "decoder_input_ids", "decoder_attention_mask", "labels"],
|
||||
)
|
||||
|
||||
output_dir = self.get_auto_remove_tmp_dir()
|
||||
|
||||
training_args = Seq2SeqTrainingArguments(
|
||||
output_dir=output_dir,
|
||||
per_device_train_batch_size=batch_size,
|
||||
per_device_eval_batch_size=batch_size,
|
||||
predict_with_generate=True,
|
||||
evaluate_during_training=True,
|
||||
do_train=True,
|
||||
do_eval=True,
|
||||
warmup_steps=0,
|
||||
eval_steps=2,
|
||||
logging_steps=2,
|
||||
)
|
||||
|
||||
# instantiate trainer
|
||||
trainer = Seq2SeqTrainer(
|
||||
model=bert2bert,
|
||||
args=training_args,
|
||||
compute_metrics=_compute_metrics,
|
||||
train_dataset=train_dataset,
|
||||
eval_dataset=val_dataset,
|
||||
)
|
||||
|
||||
# start training
|
||||
trainer.train()
|
||||
|
||||
def run_trainer(self, eval_steps: int, max_len: str, model_name: str, num_train_epochs: int):
|
||||
|
||||
# XXX: remove hardcoded path
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
---
|
||||
language: da
|
||||
tags:
|
||||
- bert
|
||||
- masked-lm
|
||||
- lm-head
|
||||
license: cc-by-4.0
|
||||
datasets:
|
||||
- common_crawl
|
||||
- wikipedia
|
||||
pipeline_tag: fill-mask
|
||||
widget:
|
||||
- text: "København er [MASK] i Danmark."
|
||||
---
|
||||
|
||||
# Danish BERT (uncased) model
|
||||
|
||||
[BotXO.ai](https://www.botxo.ai/) developed this model. For data and training details see their [GitHub repository](https://github.com/botxo/nordic_bert).
|
||||
|
||||
The original model was trained in TensorFlow then I converted it to Pytorch using [transformers-cli](https://huggingface.co/transformers/converting_tensorflow_models.html?highlight=cli).
|
||||
|
||||
For TensorFlow version download here: https://www.dropbox.com/s/19cjaoqvv2jicq9/danish_bert_uncased_v2.zip?dl=1
|
||||
|
||||
|
||||
## Architecture
|
||||
|
||||
```python
|
||||
from transformers import AutoModelForPreTraining
|
||||
|
||||
model = AutoModelForPreTraining.from_pretrained("DJSammy/bert-base-danish-uncased_BotXO,ai")
|
||||
|
||||
params = list(model.named_parameters())
|
||||
print('danish_bert_uncased_v2 has {:} different named parameters.\n'.format(len(params)))
|
||||
|
||||
print('==== Embedding Layer ====\n')
|
||||
for p in params[0:5]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== First Transformer ====\n')
|
||||
for p in params[5:21]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== Last Transformer ====\n')
|
||||
for p in params[181:197]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
print('\n==== Output Layer ====\n')
|
||||
for p in params[197:]:
|
||||
print("{:<55} {:>12}".format(p[0], str(tuple(p[1].size()))))
|
||||
|
||||
# danish_bert_uncased_v2 has 206 different named parameters.
|
||||
|
||||
# ==== Embedding Layer ====
|
||||
|
||||
# bert.embeddings.word_embeddings.weight (32000, 768)
|
||||
# bert.embeddings.position_embeddings.weight (512, 768)
|
||||
# bert.embeddings.token_type_embeddings.weight (2, 768)
|
||||
# bert.embeddings.LayerNorm.weight (768,)
|
||||
# bert.embeddings.LayerNorm.bias (768,)
|
||||
|
||||
# ==== First Transformer ====
|
||||
|
||||
# bert.encoder.layer.0.attention.self.query.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.query.bias (768,)
|
||||
# bert.encoder.layer.0.attention.self.key.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.key.bias (768,)
|
||||
# bert.encoder.layer.0.attention.self.value.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.self.value.bias (768,)
|
||||
# bert.encoder.layer.0.attention.output.dense.weight (768, 768)
|
||||
# bert.encoder.layer.0.attention.output.dense.bias (768,)
|
||||
# bert.encoder.layer.0.attention.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.0.attention.output.LayerNorm.bias (768,)
|
||||
# bert.encoder.layer.0.intermediate.dense.weight (3072, 768)
|
||||
# bert.encoder.layer.0.intermediate.dense.bias (3072,)
|
||||
# bert.encoder.layer.0.output.dense.weight (768, 3072)
|
||||
# bert.encoder.layer.0.output.dense.bias (768,)
|
||||
# bert.encoder.layer.0.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.0.output.LayerNorm.bias (768,)
|
||||
|
||||
# ==== Last Transformer ====
|
||||
|
||||
# bert.encoder.layer.11.attention.self.query.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.query.bias (768,)
|
||||
# bert.encoder.layer.11.attention.self.key.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.key.bias (768,)
|
||||
# bert.encoder.layer.11.attention.self.value.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.self.value.bias (768,)
|
||||
# bert.encoder.layer.11.attention.output.dense.weight (768, 768)
|
||||
# bert.encoder.layer.11.attention.output.dense.bias (768,)
|
||||
# bert.encoder.layer.11.attention.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.11.attention.output.LayerNorm.bias (768,)
|
||||
# bert.encoder.layer.11.intermediate.dense.weight (3072, 768)
|
||||
# bert.encoder.layer.11.intermediate.dense.bias (3072,)
|
||||
# bert.encoder.layer.11.output.dense.weight (768, 3072)
|
||||
# bert.encoder.layer.11.output.dense.bias (768,)
|
||||
# bert.encoder.layer.11.output.LayerNorm.weight (768,)
|
||||
# bert.encoder.layer.11.output.LayerNorm.bias (768,)
|
||||
|
||||
# ==== Output Layer ====
|
||||
|
||||
# bert.pooler.dense.weight (768, 768)
|
||||
# bert.pooler.dense.bias (768,)
|
||||
# cls.predictions.bias (32000,)
|
||||
# cls.predictions.transform.dense.weight (768, 768)
|
||||
# cls.predictions.transform.dense.bias (768,)
|
||||
# cls.predictions.transform.LayerNorm.weight (768,)
|
||||
# cls.predictions.transform.LayerNorm.bias (768,)
|
||||
# cls.seq_relationship.weight (2, 768)
|
||||
# cls.seq_relationship.bias (2,)
|
||||
```
|
||||
|
||||
## Example Pipeline
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
unmasker = pipeline('fill-mask', model='DJSammy/bert-base-danish-uncased_BotXO,ai')
|
||||
|
||||
unmasker('København er [MASK] i Danmark.')
|
||||
|
||||
# Copenhagen is the [MASK] of Denmark.
|
||||
# =>
|
||||
|
||||
# [{'score': 0.788068950176239,
|
||||
# 'sequence': '[CLS] københavn er hovedstad i danmark. [SEP]',
|
||||
# 'token': 12610,
|
||||
# 'token_str': 'hovedstad'},
|
||||
# {'score': 0.07606703042984009,
|
||||
# 'sequence': '[CLS] københavn er hovedstaden i danmark. [SEP]',
|
||||
# 'token': 8108,
|
||||
# 'token_str': 'hovedstaden'},
|
||||
# {'score': 0.04299738258123398,
|
||||
# 'sequence': '[CLS] københavn er metropol i danmark. [SEP]',
|
||||
# 'token': 23305,
|
||||
# 'token_str': 'metropol'},
|
||||
# {'score': 0.008163209073245525,
|
||||
# 'sequence': '[CLS] københavn er ikke i danmark. [SEP]',
|
||||
# 'token': 89,
|
||||
# 'token_str': 'ikke'},
|
||||
# {'score': 0.006238455418497324,
|
||||
# 'sequence': '[CLS] københavn er ogsa i danmark. [SEP]',
|
||||
# 'token': 25253,
|
||||
# 'token_str': 'ogsa'}]
|
||||
```
|
||||
@@ -4,44 +4,4 @@ license: mit
|
||||
---
|
||||
|
||||
# bert-german-dbmdz-uncased-sentence-stsb
|
||||
|
||||
## How to use
|
||||
**The usage description above - provided by Hugging Face - is wrong! Please use this:**
|
||||
|
||||
Install the `sentence-transformers` package. See here: <https://github.com/UKPLab/sentence-transformers>
|
||||
```python
|
||||
from sentence_transformers import models
|
||||
from sentence_transformers import SentenceTransformer
|
||||
|
||||
# load BERT model from Hugging Face
|
||||
word_embedding_model = models.Transformer(
|
||||
'T-Systems-onsite/bert-german-dbmdz-uncased-sentence-stsb')
|
||||
|
||||
# Apply mean pooling to get one fixed sized sentence vector
|
||||
pooling_model = models.Pooling(word_embedding_model.get_word_embedding_dimension(),
|
||||
pooling_mode_mean_tokens=True,
|
||||
pooling_mode_cls_token=False,
|
||||
pooling_mode_max_tokens=False)
|
||||
|
||||
# join BERT model and pooling to get the sentence transformer
|
||||
model = SentenceTransformer(modules=[word_embedding_model, pooling_model])
|
||||
```
|
||||
|
||||
## Model description
|
||||
This is a German [sentence embedding](https://github.com/UKPLab/sentence-transformers) trained on the [German STSbenchmark Dataset](https://github.com/t-systems-on-site-services-gmbh/german-STSbenchmark). It was trained from [Philip May](https://eniak.de/) and open-sourced by [T-Systems-onsite](https://www.t-systems-onsite.de/).The base language model is the [dbmdz/bert-base-german-uncased](https://huggingface.co/dbmdz/bert-base-german-uncased) from [Bayerische Staatsbibliothek ](https://huggingface.co/dbmdz).
|
||||
|
||||
## Intended uses
|
||||
> Sentence-BERT (SBERT) is a modification of the pretrained BERT network that use siamese and triplet network structures to derive semantically mean-ingful sentence embeddings that can be compared using cosine-similarity. This reduces the effort for finding the most similar pair from 65hours with BERT / RoBERTa to about 5 seconds with SBERT, while maintaining the accuracy from BERT.
|
||||
|
||||
Source: [Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks](https://arxiv.org/abs/1908.10084)
|
||||
|
||||
## Training procedure
|
||||
We did an automatic hyperprameter optimization with [Optuna](https://github.com/optuna/optuna) and found the following hyperprameters:
|
||||
- batch_size = 5
|
||||
- num_epochs = 11
|
||||
- lr = 2.637549780860126e-05
|
||||
- eps = 5.0696075038683e-06
|
||||
- weight_decay = 0.02817210102940054
|
||||
- warmup_steps = 27.342745941760147 % of total steps
|
||||
|
||||
The final model was trained on the combination of all three datasets: `sts_de_dev.csv`, `sts_de_test.csv` and `sts_de_train.csv`
|
||||
**This model is outdated! Please use this improved version: <https://huggingface.co/T-Systems-onsite/german-roberta-sentence-transformer-v2>**
|
||||
@@ -0,0 +1,66 @@
|
||||
---
|
||||
language: de
|
||||
license: mit
|
||||
---
|
||||
|
||||
# German RoBERTa for Sentence Embeddings V2
|
||||
This model is intended to [compute sentence (text embeddings)](https://www.sbert.net/docs/usage/computing_sentence_embeddings.html) for German text. These embeddings can then be compared with [cosine-similarity](https://en.wikipedia.org/wiki/Cosine_similarity) to find sentences with a similar semantic meaning. For example this can be useful for [semantic textual similarity](https://www.sbert.net/docs/usage/semantic_textual_similarity.html), [semantic search](https://www.sbert.net/docs/usage/semantic_search.html), or [paraphrase mining](https://www.sbert.net/docs/usage/paraphrase_mining.html). To do this you have to use the [Sentence Transformers Python framework](https://github.com/UKPLab/sentence-transformers).
|
||||
|
||||
> Sentence-BERT (SBERT) is a modification of the pretrained BERT network that use siamese and triplet network structures to derive semantically mean-ingful sentence embeddings that can be compared using cosine-similarity. This reduces the effort for finding the most similar pair from 65hours with BERT / RoBERTa to about 5 seconds with SBERT, while maintaining the accuracy from BERT.
|
||||
|
||||
Source: [Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks](https://arxiv.org/abs/1908.10084)
|
||||
|
||||
This model is fine-tuned from [Philip May](https://eniak.de/) and open-sourced by [T-Systems-onsite](https://www.t-systems-onsite.de/). Special thanks to [Nils Reimers](https://www.nils-reimers.de/) for your awesome open-source work, the Sentence Transformers, the models and all your help on GitHub.
|
||||
|
||||
## How to use
|
||||
**The usage description above - provided by Hugging Face - is wrong for sentence embeddings! Please use this:**
|
||||
|
||||
To use this model install the `sentence-transformers` package (see here: <https://github.com/UKPLab/sentence-transformers>).
|
||||
|
||||
```python
|
||||
from sentence_transformers import SentenceTransformer
|
||||
model = SentenceTransformer('T-Systems-onsite/german-roberta-sentence-transformer-v2')
|
||||
```
|
||||
|
||||
For details of usage and examples see here:
|
||||
- [Computing Sentence Embeddings](https://www.sbert.net/docs/usage/computing_sentence_embeddings.html)
|
||||
- [Semantic Textual Similarity](https://www.sbert.net/docs/usage/semantic_textual_similarity.html)
|
||||
- [Paraphrase Mining](https://www.sbert.net/docs/usage/paraphrase_mining.html)
|
||||
- [Semantic Search](https://www.sbert.net/docs/usage/semantic_search.html)
|
||||
- [Cross-Encoders](https://www.sbert.net/docs/usage/cross-encoder.html)
|
||||
- [Examples on GitHub](https://github.com/UKPLab/sentence-transformers/tree/master/examples/applications)
|
||||
|
||||
## Training
|
||||
The base model is [xlm-roberta-base](https://huggingface.co/xlm-roberta-base). This model has been further trained by [Nils Reimers](https://www.nils-reimers.de/) on a large scale paraphrase dataset for 50+ languages. [Nils Reimers](https://www.nils-reimers.de/) about this [on GitHub](https://github.com/UKPLab/sentence-transformers/issues/509#issuecomment-712243280):
|
||||
|
||||
>A paper is upcoming for the paraphrase models.
|
||||
>
|
||||
>These models were trained on various datasets with Millions of examples for paraphrases, mainly derived from Wikipedia edit logs, paraphrases mined from Wikipedia and SimpleWiki, paraphrases from news reports, AllNLI-entailment pairs with in-batch-negative loss etc.
|
||||
>
|
||||
>In internal tests, they perform much better than the NLI+STSb models as they have see more and broader type of training data. NLI+STSb has the issue that they are rather narrow in their domain and do not contain any domain specific words / sentences (like from chemistry, computer science, math etc.). The paraphrase models has seen plenty of sentences from various domains.
|
||||
>
|
||||
>More details with the setup, all the datasets, and a wider evaluation will follow soon.
|
||||
|
||||
The resulting model called `xlm-r-distilroberta-base-paraphrase-v1` has been released here: <https://github.com/UKPLab/sentence-transformers/releases/tag/v0.3.8>
|
||||
|
||||
Building on this cross language model we fine-tuned it for German language on the deepl.com dataset of our [German STSbenchmark dataset](https://github.com/t-systems-on-site-services-gmbh/german-STSbenchmark).
|
||||
|
||||
We did an automatic hyperprameter search for 102 trials with [Optuna](https://github.com/optuna/optuna). Using crossvalidation on the deepl.com test and dev dataset we found the following best hyperprameters:
|
||||
- batch_size = 15
|
||||
- num_epochs = 4
|
||||
- lr = 2.2995320905210864e-05
|
||||
- eps = 1.8979875906303792e-06
|
||||
- weight_decay = 0.003314045812507563
|
||||
- warmup_steps_proportion = 0.46141685205829014
|
||||
|
||||
The final model was trained with these hyperparameters on the combination of `sts_de_train.csv` and `sts_de_dev.csv`. The `sts_de_test.csv` was left for testing. The AWS dataset has not been used.
|
||||
|
||||
# Evaluation
|
||||
The evaluation has been done on the test set of our [German STSbenchmark dataset](https://github.com/t-systems-on-site-services-gmbh/german-STSbenchmark). The code is available on [Colab](https://colab.research.google.com/drive/1aCWOqDQx953kEnQ5k4Qn7uiixokocOHv?usp=sharing). As the metric for evaluation we use the Spearman’s rank correlation between the cosine-similarity of the sentence embeddings and STSbenchmark labels.
|
||||
|
||||
| Model Name | Spearman rank correlation |
|
||||
|--------------------------------------|-----------------------------------|
|
||||
| xlm-r-distilroberta-base-paraphrase-v1 | 0.8079 |
|
||||
| xlm-r-100langs-bert-base-nli-stsb-mean-tokens | 0.8194 |
|
||||
| xlm-r-bert-base-nli-stsb-mean-tokens | 0.8194 |
|
||||
| **T-Systems-onsite/german-roberta-sentence-transformer-v2** | **0.8529** |
|
||||
@@ -0,0 +1,9 @@
|
||||
# DynaBERT: Dynamic BERT with Adaptive Width and Depth
|
||||
|
||||
* DynaBERT can flexibly adjust the size and latency by selecting adaptive width and depth, and
|
||||
the subnetworks of it have competitive performances as other similar-sized compressed models.
|
||||
The training process of DynaBERT includes first training a width-adaptive BERT and then
|
||||
allowing both adaptive width and depth using knowledge distillation.
|
||||
|
||||
* This code is modified based on the repository developed by Hugging Face: [Transformers v2.1.1](https://github.com/huggingface/transformers/tree/v2.1.1)
|
||||
* The results in the paper are produced by using single V100 GPU.
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Cased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Cased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-cased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-cased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Uncased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Base Uncased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-base-uncased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Cased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Cased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-cased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-cased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Uncased Discriminator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the discriminator model, which is the main Transformer used for finetuning to downstream tasks. For generation, mask-filling, and retraining, refer to the Generator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-discriminator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
language: tl
|
||||
tags:
|
||||
- electra
|
||||
- tagalog
|
||||
- filipino
|
||||
license: gpl-3.0
|
||||
inference: false
|
||||
---
|
||||
|
||||
# ELECTRA Tagalog Small Uncased Generator
|
||||
Tagalog ELECTRA model pretrained with a large corpus scraped from the internet. This model is part of a larger research project. We open-source the model to allow greater usage within the Filipino NLP community.
|
||||
|
||||
This is the generator model used to sample synthetic text and pretrain the discriminator. Only use this model for retraining and mask-filling. For the actual model for downstream tasks, please refer to the discriminator models.
|
||||
|
||||
## Usage
|
||||
The model can be loaded and used in both PyTorch and TensorFlow through the HuggingFace Transformers package.
|
||||
|
||||
```python
|
||||
from transformers import TFAutoModel, AutoModel, AutoTokenizer
|
||||
|
||||
# TensorFlow
|
||||
model = TFAutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator', from_pt=True)
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator', do_lower_case=False)
|
||||
|
||||
# PyTorch
|
||||
model = AutoModel.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator')
|
||||
tokenizer = AutoTokenizer.from_pretrained('jcblaise/electra-tagalog-small-uncased-generator', do_lower_case=False)
|
||||
```
|
||||
Finetuning scripts and other utilities we use for our projects can be found in our centralized repository at https://github.com/jcblaisecruz02/Filipino-Text-Benchmarks
|
||||
|
||||
## Citations
|
||||
All model details and training setups can be found in our papers. If you use our model or find it useful in your projects, please cite our work:
|
||||
|
||||
```
|
||||
@article{cruz2020investigating,
|
||||
title={Investigating the True Performance of Transformers in Low-Resource Languages: A Case Study in Automatic Corpus Creation},
|
||||
author={Jan Christian Blaise Cruz and Jose Kristian Resabal and James Lin and Dan John Velasco and Charibeth Cheng},
|
||||
journal={arXiv preprint arXiv:2010.11574},
|
||||
year={2020}
|
||||
}
|
||||
```
|
||||
|
||||
## Data and Other Resources
|
||||
Data used to train this model as well as other benchmark datasets in Filipino can be found in my website at https://blaisecruz.com
|
||||
|
||||
## Contact
|
||||
If you have questions, concerns, or if you just want to chat about NLP and low-resource languages in general, you may reach me through my work email at jan_christian_cruz@dlsu.edu.ph
|
||||
@@ -0,0 +1,96 @@
|
||||
---
|
||||
language: it
|
||||
datasets:
|
||||
- xtreme
|
||||
---
|
||||
|
||||
# Italian-Bert (Italian Bert) + POS 🎃🏷
|
||||
|
||||
This model is a fine-tuned on [xtreme udpos Italian](https://huggingface.co/nlp/viewer/?dataset=xtreme&config=udpos.Italian) version of [Bert Base Italian](https://huggingface.co/dbmdz/bert-base-italian-cased) for **POS** downstream task.
|
||||
|
||||
## Details of the downstream task (POS) - Dataset
|
||||
|
||||
- [Dataset: xtreme udpos Italian](https://huggingface.co/nlp/viewer/?dataset=xtreme&config=udpos.Italian) 📚
|
||||
|
||||
| Dataset | # Examples |
|
||||
| ---------------------- | ----- |
|
||||
| Train | 716 K |
|
||||
| Dev | 85 K |
|
||||
|
||||
- [Fine-tune on NER script provided by @stefan-it](https://raw.githubusercontent.com/stefan-it/fine-tuned-berts-seq/master/scripts/preprocess.py)
|
||||
|
||||
- Labels covered:
|
||||
|
||||
```
|
||||
ADJ
|
||||
ADP
|
||||
ADV
|
||||
AUX
|
||||
CCONJ
|
||||
DET
|
||||
INTJ
|
||||
NOUN
|
||||
NUM
|
||||
PART
|
||||
PRON
|
||||
PROPN
|
||||
PUNCT
|
||||
SCONJ
|
||||
SYM
|
||||
VERB
|
||||
X
|
||||
```
|
||||
|
||||
## Metrics on evaluation set 🧾
|
||||
|
||||
| Metric | # score |
|
||||
| :------------------------------------------------------------------------------------: | :-------: |
|
||||
| F1 | **97.25**
|
||||
| Precision | **97.15** |
|
||||
| Recall | **97.36** |
|
||||
|
||||
## Model in action 🔨
|
||||
|
||||
|
||||
Example of usage
|
||||
|
||||
```python
|
||||
from transformers import pipeline
|
||||
|
||||
nlp_pos = pipeline(
|
||||
"ner",
|
||||
model="sachaarbonel/bert-italian-cased-finetuned-pos",
|
||||
tokenizer=(
|
||||
'sachaarbonel/bert-spanish-cased-finetuned-pos',
|
||||
{"use_fast": False}
|
||||
))
|
||||
|
||||
|
||||
text = 'Roma è la Capitale d'Italia.'
|
||||
|
||||
nlp_pos(text)
|
||||
|
||||
'''
|
||||
Output:
|
||||
--------
|
||||
[{'entity': 'PROPN', 'index': 1, 'score': 0.9995346665382385, 'word': 'roma'},
|
||||
{'entity': 'AUX', 'index': 2, 'score': 0.9966597557067871, 'word': 'e'},
|
||||
{'entity': 'DET', 'index': 3, 'score': 0.9994786977767944, 'word': 'la'},
|
||||
{'entity': 'NOUN',
|
||||
'index': 4,
|
||||
'score': 0.9995198249816895,
|
||||
'word': 'capitale'},
|
||||
{'entity': 'ADP', 'index': 5, 'score': 0.9990894198417664, 'word': 'd'},
|
||||
{'entity': 'PART', 'index': 6, 'score': 0.57159024477005, 'word': "'"},
|
||||
{'entity': 'PROPN',
|
||||
'index': 7,
|
||||
'score': 0.9994804263114929,
|
||||
'word': 'italia'},
|
||||
{'entity': 'PUNCT', 'index': 8, 'score': 0.9772886633872986, 'word': '.'}]
|
||||
'''
|
||||
```
|
||||
Yeah! Not too bad 🎉
|
||||
|
||||
> Created by [Sacha Arbonel/@sachaarbonel](https://twitter.com/sachaarbonel) | [LinkedIn](https://www.linkedin.com/in/sacha-arbonel)
|
||||
|
||||
> Made with <span style="color: #e25555;">♥</span> in Paris
|
||||
@@ -0,0 +1,82 @@
|
||||
---
|
||||
datasets:
|
||||
- snli
|
||||
- anli
|
||||
- multi_nli
|
||||
- multi_nli_mismatch
|
||||
- fever
|
||||
license: mit
|
||||
---
|
||||
This is a strong pre-trained RoBERTa-Large NLI model.
|
||||
|
||||
The training data is a combination of well-known NLI datasets: [`SNLI`](https://nlp.stanford.edu/projects/snli/), [`MNLI`](https://cims.nyu.edu/~sbowman/multinli/), [`FEVER-NLI`](https://github.com/easonnie/combine-FEVER-NSMN/blob/master/other_resources/nli_fever.md), [`ANLI (R1, R2, R3)`](https://github.com/facebookresearch/anli).
|
||||
Other pre-trained NLI models including `RoBERTa`, `ALBert`, `BART`, `ELECTRA`, `XLNet` are also available.
|
||||
|
||||
Trained by [Yixin Nie](https://easonnie.github.io), [original source](https://github.com/facebookresearch/anli).
|
||||
|
||||
Try the code snippet below.
|
||||
```
|
||||
from transformers import AutoTokenizer, AutoModelForSequenceClassification
|
||||
import torch
|
||||
|
||||
if __name__ == '__main__':
|
||||
max_length = 256
|
||||
|
||||
premise = "Two women are embracing while holding to go packages."
|
||||
hypothesis = "The men are fighting outside a deli."
|
||||
|
||||
hg_model_hub_name = "ynie/roberta-large-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/albert-xxlarge-v2-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/bart-large-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/electra-large-discriminator-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
# hg_model_hub_name = "ynie/xlnet-large-cased-snli_mnli_fever_anli_R1_R2_R3-nli"
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(hg_model_hub_name)
|
||||
model = AutoModelForSequenceClassification.from_pretrained(hg_model_hub_name)
|
||||
|
||||
tokenized_input_seq_pair = tokenizer.encode_plus(premise, hypothesis,
|
||||
max_length=max_length,
|
||||
return_token_type_ids=True, truncation=True)
|
||||
|
||||
input_ids = torch.Tensor(tokenized_input_seq_pair['input_ids']).long().unsqueeze(0)
|
||||
# remember bart doesn't have 'token_type_ids', remove the line below if you are using bart.
|
||||
token_type_ids = torch.Tensor(tokenized_input_seq_pair['token_type_ids']).long().unsqueeze(0)
|
||||
attention_mask = torch.Tensor(tokenized_input_seq_pair['attention_mask']).long().unsqueeze(0)
|
||||
|
||||
outputs = model(input_ids,
|
||||
attention_mask=attention_mask,
|
||||
token_type_ids=token_type_ids,
|
||||
labels=None)
|
||||
# Note:
|
||||
# "id2label": {
|
||||
# "0": "entailment",
|
||||
# "1": "neutral",
|
||||
# "2": "contradiction"
|
||||
# },
|
||||
|
||||
predicted_probability = torch.softmax(outputs[0], dim=1)[0].tolist() # batch_size only one
|
||||
|
||||
print("Premise:", premise)
|
||||
print("Hypothesis:", hypothesis)
|
||||
print("Entailment:", predicted_probability[0])
|
||||
print("Neutral:", predicted_probability[1])
|
||||
print("Contradiction:", predicted_probability[2])
|
||||
```
|
||||
|
||||
More in [here](https://github.com/facebookresearch/anli/blob/master/src/hg_api/interactive_eval.py).
|
||||
|
||||
Citation:
|
||||
```
|
||||
@inproceedings{nie-etal-2020-adversarial,
|
||||
title = "Adversarial {NLI}: A New Benchmark for Natural Language Understanding",
|
||||
author = "Nie, Yixin and
|
||||
Williams, Adina and
|
||||
Dinan, Emily and
|
||||
Bansal, Mohit and
|
||||
Weston, Jason and
|
||||
Kiela, Douwe",
|
||||
booktitle = "Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics",
|
||||
year = "2020",
|
||||
publisher = "Association for Computational Linguistics",
|
||||
}
|
||||
```
|
||||
@@ -115,6 +115,11 @@ class RobertaEmbeddings(nn.Module):
|
||||
|
||||
if inputs_embeds is None:
|
||||
inputs_embeds = self.word_embeddings(input_ids)
|
||||
|
||||
max_position_embeddings = self.position_embeddings.num_embeddings
|
||||
if position_ids.max() > max_position_embeddings:
|
||||
raise ValueError("Position ids are too large, the max is {}.".format(max_position_embeddings))
|
||||
|
||||
position_embeddings = self.position_embeddings(position_ids)
|
||||
token_type_embeddings = self.token_type_embeddings(token_type_ids)
|
||||
|
||||
|
||||
@@ -129,6 +129,9 @@ class AlbertTokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
super().__init__(
|
||||
do_lower_case=do_lower_case,
|
||||
remove_space=remove_space,
|
||||
keep_accents=keep_accents,
|
||||
bos_token=bos_token,
|
||||
eos_token=eos_token,
|
||||
unk_token=unk_token,
|
||||
|
||||
@@ -122,7 +122,7 @@ class BartTokenizer(RobertaTokenizer):
|
||||
- **attention_mask** -- List of indices specifying which tokens should be attended to by the model.
|
||||
- **labels** -- List of token ids for tgt_texts
|
||||
|
||||
The full set of keys ``[input_ids, attention_mask, decoder_input_ids, decoder_attention_mask]``,
|
||||
The full set of keys ``[input_ids, attention_mask, labels]``,
|
||||
will only be returned if tgt_texts is passed. Otherwise, input_ids, attention_mask will be the only keys.
|
||||
"""
|
||||
kwargs.pop("src_lang", None)
|
||||
|
||||
@@ -178,11 +178,16 @@ class BertTokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
super().__init__(
|
||||
do_lower_case=do_lower_case,
|
||||
do_basic_tokenize=do_basic_tokenize,
|
||||
never_split=never_split,
|
||||
unk_token=unk_token,
|
||||
sep_token=sep_token,
|
||||
pad_token=pad_token,
|
||||
cls_token=cls_token,
|
||||
mask_token=mask_token,
|
||||
tokenize_chinese_chars=tokenize_chinese_chars,
|
||||
strip_accents=strip_accents,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
@@ -129,11 +129,12 @@ class BertweetTokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
super().__init__(
|
||||
normalization=normalization,
|
||||
bos_token=bos_token,
|
||||
eos_token=eos_token,
|
||||
unk_token=unk_token,
|
||||
sep_token=sep_token,
|
||||
cls_token=cls_token,
|
||||
unk_token=unk_token,
|
||||
pad_token=pad_token,
|
||||
mask_token=mask_token,
|
||||
**kwargs,
|
||||
|
||||
@@ -308,16 +308,13 @@ class GPT2Tokenizer(object):
|
||||
|
||||
- We remapped the token ids in our dictionary with regarding to the new special tokens, `[PAD]` => 0, `[CLS]` => 1, `[SEP]` => 2, `[UNK]` => 3, `[MASK]` => 50264
|
||||
|
||||
do_lower_case (:obj:`bool`, optional):
|
||||
Whether to convert inputs to lower case. **Not used in GPT2 tokenizer**.
|
||||
|
||||
special_tokens (:obj:`list`, optional):
|
||||
List of special tokens to be added to the end of the vocabulary.
|
||||
|
||||
|
||||
"""
|
||||
|
||||
def __init__(self, vocab_file=None, do_lower_case=True, special_tokens=None):
|
||||
def __init__(self, vocab_file=None, special_tokens=None):
|
||||
self.pad_token = "[PAD]"
|
||||
self.sep_token = "[SEP]"
|
||||
self.unk_token = "[UNK]"
|
||||
@@ -523,6 +520,7 @@ class DebertaTokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
super().__init__(
|
||||
do_lower_case=do_lower_case,
|
||||
unk_token=unk_token,
|
||||
sep_token=sep_token,
|
||||
pad_token=pad_token,
|
||||
|
||||
@@ -194,6 +194,9 @@ class FSMTTokenizer(PreTrainedTokenizer):
|
||||
):
|
||||
super().__init__(
|
||||
langs=langs,
|
||||
src_vocab_file=src_vocab_file,
|
||||
tgt_vocab_file=tgt_vocab_file,
|
||||
merges_file=merges_file,
|
||||
unk_token=unk_token,
|
||||
bos_token=bos_token,
|
||||
sep_token=sep_token,
|
||||
|
||||
@@ -164,7 +164,14 @@ class GPT2Tokenizer(PreTrainedTokenizer):
|
||||
bos_token = AddedToken(bos_token, lstrip=False, rstrip=False) if isinstance(bos_token, str) else bos_token
|
||||
eos_token = AddedToken(eos_token, lstrip=False, rstrip=False) if isinstance(eos_token, str) else eos_token
|
||||
unk_token = AddedToken(unk_token, lstrip=False, rstrip=False) if isinstance(unk_token, str) else unk_token
|
||||
super().__init__(bos_token=bos_token, eos_token=eos_token, unk_token=unk_token, **kwargs)
|
||||
super().__init__(
|
||||
errors=errors,
|
||||
unk_token=unk_token,
|
||||
bos_token=bos_token,
|
||||
eos_token=eos_token,
|
||||
add_prefix_space=add_prefix_space,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
with open(vocab_file, encoding="utf-8") as vocab_handle:
|
||||
self.encoder = json.load(vocab_handle)
|
||||
|
||||
@@ -97,10 +97,12 @@ class MarianTokenizer(PreTrainedTokenizer):
|
||||
):
|
||||
super().__init__(
|
||||
# bos_token=bos_token, unused. Start decoding with config.decoder_start_token_id
|
||||
model_max_length=model_max_length,
|
||||
eos_token=eos_token,
|
||||
source_lang=source_lang,
|
||||
target_lang=target_lang,
|
||||
unk_token=unk_token,
|
||||
eos_token=eos_token,
|
||||
pad_token=pad_token,
|
||||
model_max_length=model_max_length,
|
||||
**kwargs,
|
||||
)
|
||||
assert Path(source_spm).exists(), f"cannot find spm source {source_spm}"
|
||||
|
||||
@@ -47,8 +47,8 @@ class PegasusTokenizer(ReformerTokenizer):
|
||||
pretrained_vocab_files_map = PRETRAINED_VOCAB_FILES_MAP
|
||||
max_model_input_sizes = PRETRAINED_POSITIONAL_EMBEDDINGS_SIZES
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
def __init__(self, *args, pad_token="<pad>", **kwargs):
|
||||
super().__init__(*args, **kwargs, pad_token="<pad>")
|
||||
# Don't use reserved words added_token_encoder, added_tokens_decoder because of
|
||||
# AssertionError: Non-consecutive added token '1' found. in from_pretrained
|
||||
assert len(self.added_tokens_decoder) == 0
|
||||
|
||||
@@ -119,11 +119,16 @@ class ProphetNetTokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
super().__init__(
|
||||
do_lower_case=do_lower_case,
|
||||
do_basic_tokenize=do_basic_tokenize,
|
||||
never_split=never_split,
|
||||
unk_token=unk_token,
|
||||
sep_token=sep_token,
|
||||
x_sep_token=x_sep_token,
|
||||
pad_token=pad_token,
|
||||
mask_token=mask_token,
|
||||
x_sep_token=x_sep_token,
|
||||
tokenize_chinese_chars=tokenize_chinese_chars,
|
||||
strip_accents=strip_accents,
|
||||
**kwargs,
|
||||
)
|
||||
self.unique_no_split_tokens.append(x_sep_token)
|
||||
|
||||
@@ -86,19 +86,10 @@ class ReformerTokenizer(PreTrainedTokenizer):
|
||||
max_model_input_sizes = PRETRAINED_POSITIONAL_EMBEDDINGS_SIZES
|
||||
model_input_names = ["attention_mask"]
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
vocab_file,
|
||||
eos_token="</s>",
|
||||
unk_token="<unk>",
|
||||
pad_token="<pad>",
|
||||
additional_special_tokens=[],
|
||||
**kwargs
|
||||
):
|
||||
def __init__(self, vocab_file, eos_token="</s>", unk_token="<unk>", additional_special_tokens=[], **kwargs):
|
||||
super().__init__(
|
||||
eos_token=eos_token,
|
||||
unk_token=unk_token,
|
||||
pad_token=pad_token,
|
||||
additional_special_tokens=additional_special_tokens,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
@@ -102,7 +102,6 @@ class ReformerTokenizerFast(PreTrainedTokenizerFast):
|
||||
tokenizer_file=None,
|
||||
eos_token="</s>",
|
||||
unk_token="<unk>",
|
||||
pad_token="<pad>",
|
||||
additional_special_tokens=[],
|
||||
**kwargs
|
||||
):
|
||||
@@ -111,7 +110,6 @@ class ReformerTokenizerFast(PreTrainedTokenizerFast):
|
||||
tokenizer_file=tokenizer_file,
|
||||
eos_token=eos_token,
|
||||
unk_token=unk_token,
|
||||
pad_token=pad_token,
|
||||
additional_special_tokens=additional_special_tokens,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
@@ -112,15 +112,22 @@ class T5Tokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
# Add extra_ids to the special token list
|
||||
if extra_ids > 0:
|
||||
if additional_special_tokens is None:
|
||||
additional_special_tokens = []
|
||||
additional_special_tokens.extend(["<extra_id_{}>".format(i) for i in range(extra_ids)])
|
||||
if extra_ids > 0 and additional_special_tokens is None:
|
||||
additional_special_tokens = ["<extra_id_{}>".format(i) for i in range(extra_ids)]
|
||||
elif extra_ids > 0 and additional_special_tokens is not None:
|
||||
# Check that we have the right number of extra_id special tokens
|
||||
extra_tokens = len(set(filter(lambda x: bool("extra_id" in x), additional_special_tokens)))
|
||||
if extra_tokens != extra_ids:
|
||||
raise ValueError(
|
||||
f"Both extra_ids ({extra_ids}) and additional_special_tokens ({additional_special_tokens}) are provided to T5Tokenizer. "
|
||||
"In this case the additional_special_tokens must include the extra_ids tokens"
|
||||
)
|
||||
|
||||
super().__init__(
|
||||
eos_token=eos_token,
|
||||
unk_token=unk_token,
|
||||
pad_token=pad_token,
|
||||
extra_ids=extra_ids,
|
||||
additional_special_tokens=additional_special_tokens,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
@@ -126,6 +126,18 @@ class T5TokenizerFast(PreTrainedTokenizerFast):
|
||||
additional_special_tokens=None,
|
||||
**kwargs
|
||||
):
|
||||
# Add extra_ids to the special token list
|
||||
if extra_ids > 0 and additional_special_tokens is None:
|
||||
additional_special_tokens = ["<extra_id_{}>".format(i) for i in range(extra_ids)]
|
||||
elif extra_ids > 0 and additional_special_tokens is not None:
|
||||
# Check that we have the right number of extra special tokens
|
||||
extra_tokens = len(set(filter(lambda x: bool("extra_id_" in x), additional_special_tokens)))
|
||||
if extra_tokens != extra_ids:
|
||||
raise ValueError(
|
||||
f"Both extra_ids ({extra_ids}) and additional_special_tokens ({additional_special_tokens}) are provided to T5Tokenizer. "
|
||||
"In this case the additional_special_tokens must include the extra_ids tokens"
|
||||
)
|
||||
|
||||
super().__init__(
|
||||
vocab_file,
|
||||
tokenizer_file=tokenizer_file,
|
||||
@@ -137,13 +149,6 @@ class T5TokenizerFast(PreTrainedTokenizerFast):
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
if extra_ids > 0:
|
||||
all_extra_tokens = ["<extra_id_{}>".format(i) for i in range(extra_ids)]
|
||||
if all(tok not in self.additional_special_tokens for tok in all_extra_tokens):
|
||||
self.additional_special_tokens = self.additional_special_tokens + [
|
||||
"<extra_id_{}>".format(i) for i in range(extra_ids)
|
||||
]
|
||||
|
||||
self.vocab_file = vocab_file
|
||||
self._extra_ids = extra_ids
|
||||
|
||||
|
||||
@@ -164,7 +164,19 @@ class TransfoXLTokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
super().__init__(
|
||||
unk_token=unk_token, eos_token=eos_token, additional_special_tokens=additional_special_tokens, **kwargs
|
||||
special=special,
|
||||
min_freq=min_freq,
|
||||
max_size=max_size,
|
||||
lower_case=lower_case,
|
||||
delimiter=delimiter,
|
||||
vocab_file=vocab_file,
|
||||
pretrained_vocab_file=pretrained_vocab_file,
|
||||
never_split=never_split,
|
||||
unk_token=unk_token,
|
||||
eos_token=eos_token,
|
||||
additional_special_tokens=additional_special_tokens,
|
||||
language=language,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
if never_split is None:
|
||||
|
||||
@@ -364,7 +364,7 @@ class BatchEncoding(UserDict):
|
||||
token_index = self._seq_len + token_index
|
||||
return self._encodings[batch_index].token_to_word(token_index)
|
||||
|
||||
def word_to_tokens(self, batch_or_word_index: int, word_index: Optional[int] = None) -> TokenSpan:
|
||||
def word_to_tokens(self, batch_or_word_index: int, word_index: Optional[int] = None) -> Optional[TokenSpan]:
|
||||
"""
|
||||
Get the encoded token span corresponding to a word in the sequence of the batch.
|
||||
|
||||
@@ -391,8 +391,9 @@ class BatchEncoding(UserDict):
|
||||
of the word in the sequence.
|
||||
|
||||
Returns:
|
||||
:class:`~transformers.tokenization_utils_base.TokenSpan`
|
||||
Span of tokens in the encoded sequence.
|
||||
Optional :class:`~transformers.tokenization_utils_base.TokenSpan`
|
||||
Span of tokens in the encoded sequence. Returns :obj:`None` if no tokens correspond
|
||||
to the word.
|
||||
"""
|
||||
|
||||
if not self._encodings:
|
||||
@@ -406,7 +407,8 @@ class BatchEncoding(UserDict):
|
||||
batch_index = self._batch_size + batch_index
|
||||
if word_index < 0:
|
||||
word_index = self._seq_len + word_index
|
||||
return TokenSpan(*(self._encodings[batch_index].word_to_tokens(word_index)))
|
||||
span = self._encodings[batch_index].word_to_tokens(word_index)
|
||||
return TokenSpan(*span) if span is not None else None
|
||||
|
||||
def token_to_chars(self, batch_or_token_index: int, token_index: Optional[int] = None) -> CharSpan:
|
||||
"""
|
||||
@@ -1362,11 +1364,9 @@ PREPARE_SEQ2SEQ_BATCH_DOCSTRING = """
|
||||
|
||||
- **input_ids** -- List of token ids to be fed to the encoder.
|
||||
- **attention_mask** -- List of indices specifying which tokens should be attended to by the model.
|
||||
- **decoder_input_ids** -- List of token ids to be fed to the decoder.
|
||||
- **decoder_attention_mask** -- List of indices specifying which tokens should be attended to by the decoder.
|
||||
This does not include causal mask, which is built by the model.
|
||||
- **labels** -- List of token ids for tgt_texts.
|
||||
|
||||
The full set of keys ``[input_ids, attention_mask, decoder_input_ids, decoder_attention_mask]``,
|
||||
The full set of keys ``[input_ids, attention_mask, labels]``,
|
||||
will only be returned if tgt_texts is passed. Otherwise, input_ids, attention_mask will be the only keys.
|
||||
|
||||
"""
|
||||
@@ -1673,7 +1673,7 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
if (
|
||||
"tokenizer_file" not in resolved_vocab_files or resolved_vocab_files["tokenizer_file"] is None
|
||||
) and cls.slow_tokenizer_class is not None:
|
||||
slow_tokenizer = cls.slow_tokenizer_class._from_pretrained(
|
||||
slow_tokenizer = (cls.slow_tokenizer_class)._from_pretrained(
|
||||
copy.deepcopy(resolved_vocab_files),
|
||||
pretrained_model_name_or_path,
|
||||
copy.deepcopy(init_configuration),
|
||||
|
||||
@@ -16,7 +16,6 @@
|
||||
For slow (python) tokenizers see tokenization_utils.py
|
||||
"""
|
||||
|
||||
import copy
|
||||
import json
|
||||
import os
|
||||
import warnings
|
||||
@@ -105,7 +104,7 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
self._tokenizer = fast_tokenizer
|
||||
|
||||
if slow_tokenizer is not None:
|
||||
kwargs = copy.deepcopy(slow_tokenizer.init_kwargs)
|
||||
kwargs.update(slow_tokenizer.init_kwargs)
|
||||
|
||||
# We call this after having initialized the backend tokenizer because we update it.
|
||||
super().__init__(**kwargs)
|
||||
|
||||
@@ -621,6 +621,9 @@ class XLMTokenizer(PreTrainedTokenizer):
|
||||
cls_token=cls_token,
|
||||
mask_token=mask_token,
|
||||
additional_special_tokens=additional_special_tokens,
|
||||
lang2id=lang2id,
|
||||
id2lang=id2lang,
|
||||
do_lowercase_and_remove_accent=do_lowercase_and_remove_accent,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
@@ -123,9 +123,10 @@ class XLMProphetNetTokenizer(PreTrainedTokenizer):
|
||||
super().__init__(
|
||||
bos_token=bos_token,
|
||||
eos_token=eos_token,
|
||||
unk_token=unk_token,
|
||||
sep_token=sep_token,
|
||||
unk_token=unk_token,
|
||||
pad_token=pad_token,
|
||||
cls_token=cls_token,
|
||||
mask_token=mask_token,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
@@ -128,6 +128,9 @@ class XLNetTokenizer(PreTrainedTokenizer):
|
||||
**kwargs
|
||||
):
|
||||
super().__init__(
|
||||
do_lower_case=do_lower_case,
|
||||
remove_space=remove_space,
|
||||
keep_accents=keep_accents,
|
||||
bos_token=bos_token,
|
||||
eos_token=eos_token,
|
||||
unk_token=unk_token,
|
||||
|
||||
@@ -371,6 +371,20 @@ class RobertaModelTest(ModelTesterMixin, unittest.TestCase):
|
||||
self.assertEqual(position_ids.shape, expected_positions.shape)
|
||||
self.assertTrue(torch.all(torch.eq(position_ids, expected_positions)))
|
||||
|
||||
def test_handling_too_long_sequences_for_position_ids(self):
|
||||
config = self.model_tester.prepare_config_and_inputs()[0]
|
||||
model = RobertaEmbeddings(config=config)
|
||||
|
||||
input_ids = torch.zeros((1, config.max_position_embeddings + 1)).long()
|
||||
expected_positions = torch.as_tensor([list(range(config.max_position_embeddings + 1))]) + model.padding_idx + 1
|
||||
|
||||
position_ids = create_position_ids_from_input_ids(input_ids, model.padding_idx)
|
||||
self.assertEqual(position_ids.shape, expected_positions.shape)
|
||||
self.assertTrue(torch.all(torch.eq(position_ids, expected_positions)))
|
||||
|
||||
with self.assertRaises(ValueError):
|
||||
model.forward(input_ids)
|
||||
|
||||
def test_create_position_ids_from_inputs_embeds(self):
|
||||
"""Ensure that the default position ids only assign a sequential . This is a regression
|
||||
test for https://github.com/huggingface/transformers/issues/1761
|
||||
|
||||
@@ -3,7 +3,7 @@ from typing import List, Optional
|
||||
|
||||
from transformers import is_tf_available, is_torch_available, pipeline
|
||||
from transformers.pipelines import DefaultArgumentHandler, Pipeline
|
||||
from transformers.testing_utils import _run_slow_tests, is_pipeline_test, require_tf, require_tokenizers, require_torch, slow
|
||||
from transformers.testing_utils import _run_slow_tests, is_pipeline_test, require_tf, require_torch, slow
|
||||
|
||||
|
||||
VALID_INPUTS = ["A simple string", ["list of strings"]]
|
||||
@@ -79,26 +79,9 @@ class CustomInputPipelineCommonMixin:
|
||||
nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="tf")
|
||||
self._test_pipeline(nlp)
|
||||
|
||||
@require_torch
|
||||
def test_fast_tokenizer_equivalence_torch_small(self):
|
||||
for model_name in self.small_models:
|
||||
fast_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="pt", use_fast=True)
|
||||
slow_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="pt", use_fast=False)
|
||||
self._test_equivalence(fast_nlp, slow_nlp)
|
||||
|
||||
@require_tf
|
||||
def test_fast_tokenizer_equivalence_tf_small(self):
|
||||
for model_name in self.small_models:
|
||||
fast_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="tf", use_fast=True)
|
||||
slow_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="tf", use_fast=False)
|
||||
self._test_equivalence(fast_nlp, slow_nlp)
|
||||
|
||||
def _test_pipeline(self, nlp: Pipeline):
|
||||
raise NotImplementedError
|
||||
|
||||
def _test_equivalence(self, fast_nlp: Pipeline, slow_nlp: Pipeline):
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
@is_pipeline_test
|
||||
class MonoInputPipelineCommonMixin:
|
||||
@@ -182,45 +165,6 @@ class MonoInputPipelineCommonMixin:
|
||||
)
|
||||
self._test_pipeline(nlp)
|
||||
|
||||
@require_tokenizers
|
||||
@require_torch
|
||||
def test_fast_tokenizer_equivalence_torch_small(self):
|
||||
for model_name in self.small_models:
|
||||
fast_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="pt", use_fast=True, **self.pipeline_loading_kwargs)
|
||||
slow_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="pt", use_fast=False, **self.pipeline_loading_kwargs)
|
||||
self._test_equivalence(fast_nlp, slow_nlp)
|
||||
|
||||
@require_tokenizers
|
||||
@require_tf
|
||||
def test_fast_tokenizer_equivalence_tf_small(self):
|
||||
for model_name in self.small_models:
|
||||
fast_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="tf", use_fast=True)
|
||||
slow_nlp = pipeline(task=self.pipeline_task, model=model_name, tokenizer=model_name, framework="tf", use_fast=False)
|
||||
self._test_equivalence(fast_nlp, slow_nlp)
|
||||
|
||||
def _test_equivalence(self, fast_nlp: Pipeline, slow_nlp: Pipeline):
|
||||
mono_result_fast = fast_nlp(self.valid_inputs[0], **self.pipeline_running_kwargs)
|
||||
mono_result_slow = slow_nlp(self.valid_inputs[0], **self.pipeline_running_kwargs)
|
||||
|
||||
if isinstance(mono_result_fast[0], list):
|
||||
mono_result_fast = mono_result_fast[0]
|
||||
mono_result_slow = mono_result_slow[0]
|
||||
|
||||
for key in mono_result_fast[0]:
|
||||
self.assertEqual(mono_result_fast[0][key], mono_result_slow[0][key])
|
||||
|
||||
multi_result_fast = [fast_nlp(input, **self.pipeline_running_kwargs) for input in self.valid_inputs]
|
||||
multi_result_slow = [slow_nlp(input, **self.pipeline_running_kwargs) for input in self.valid_inputs]
|
||||
|
||||
if self.expected_multi_result is not None:
|
||||
for result_fast, result_slow, expect in zip(multi_result_fast, multi_result_slow, self.expected_multi_result):
|
||||
for key in result_fast:
|
||||
self.assertEqual(
|
||||
set([o[key] for o in result_fast]),
|
||||
set([o[key] for o in result_slow]),
|
||||
set([o[key] for o in expect]),
|
||||
)
|
||||
|
||||
def _test_pipeline(self, nlp: Pipeline):
|
||||
self.assertIsNotNone(nlp)
|
||||
|
||||
|
||||
@@ -177,6 +177,25 @@ class TokenizerTesterMixin:
|
||||
self.assertIn("tokenizer_file", signature.parameters)
|
||||
self.assertIsNone(signature.parameters["tokenizer_file"].default)
|
||||
|
||||
def test_tokenizer_slow_store_full_signature(self):
|
||||
signature = inspect.signature(self.tokenizer_class.__init__)
|
||||
tokenizer = self.get_tokenizer()
|
||||
|
||||
for parameter_name, parameter in signature.parameters.items():
|
||||
if parameter.default != inspect.Parameter.empty:
|
||||
self.assertIn(parameter_name, tokenizer.init_kwargs)
|
||||
|
||||
def test_tokenizer_fast_store_full_signature(self):
|
||||
if not self.test_rust_tokenizer:
|
||||
return
|
||||
|
||||
signature = inspect.signature(self.rust_tokenizer_class.__init__)
|
||||
tokenizer = self.get_rust_tokenizer()
|
||||
|
||||
for parameter_name, parameter in signature.parameters.items():
|
||||
if parameter.default != inspect.Parameter.empty:
|
||||
self.assertIn(parameter_name, tokenizer.init_kwargs)
|
||||
|
||||
def test_rust_and_python_full_tokenizers(self):
|
||||
if not self.test_rust_tokenizer:
|
||||
return
|
||||
|
||||
@@ -63,6 +63,50 @@ class ReformerTokenizationTest(TokenizerTesterMixin, unittest.TestCase):
|
||||
rust_ids = rust_tokenizer.encode(sequence)
|
||||
self.assertListEqual(ids, rust_ids)
|
||||
|
||||
def test_padding(self, max_length=15):
|
||||
for tokenizer, pretrained_name, kwargs in self.tokenizers_list:
|
||||
with self.subTest("{} ({})".format(tokenizer.__class__.__name__, pretrained_name)):
|
||||
tokenizer_r = self.rust_tokenizer_class.from_pretrained(pretrained_name, **kwargs)
|
||||
|
||||
# Simple input
|
||||
s = "This is a simple input"
|
||||
s2 = ["This is a simple input 1", "This is a simple input 2"]
|
||||
p = ("This is a simple input", "This is a pair")
|
||||
p2 = [
|
||||
("This is a simple input 1", "This is a simple input 2"),
|
||||
("This is a simple pair 1", "This is a simple pair 2"),
|
||||
]
|
||||
|
||||
# Simple input tests
|
||||
self.assertRaises(ValueError, tokenizer_r.encode, s, max_length=max_length, padding="max_length")
|
||||
|
||||
# Simple input
|
||||
self.assertRaises(ValueError, tokenizer_r.encode_plus, s, max_length=max_length, padding="max_length")
|
||||
|
||||
# Simple input
|
||||
self.assertRaises(
|
||||
ValueError,
|
||||
tokenizer_r.batch_encode_plus,
|
||||
s2,
|
||||
max_length=max_length,
|
||||
padding="max_length",
|
||||
)
|
||||
|
||||
# Pair input
|
||||
self.assertRaises(ValueError, tokenizer_r.encode, p, max_length=max_length, padding="max_length")
|
||||
|
||||
# Pair input
|
||||
self.assertRaises(ValueError, tokenizer_r.encode_plus, p, max_length=max_length, padding="max_length")
|
||||
|
||||
# Pair input
|
||||
self.assertRaises(
|
||||
ValueError,
|
||||
tokenizer_r.batch_encode_plus,
|
||||
p2,
|
||||
max_length=max_length,
|
||||
padding="max_length",
|
||||
)
|
||||
|
||||
def test_full_tokenizer(self):
|
||||
tokenizer = ReformerTokenizer(SAMPLE_VOCAB, keep_accents=True)
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ from typing import Callable, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
from transformers import BatchEncoding, BertTokenizer, BertTokenizerFast, PreTrainedTokenizer, TensorType
|
||||
from transformers import BatchEncoding, BertTokenizer, BertTokenizerFast, PreTrainedTokenizer, TensorType, TokenSpan
|
||||
from transformers.testing_utils import require_tf, require_tokenizers, require_torch, slow
|
||||
from transformers.tokenization_gpt2 import GPT2Tokenizer
|
||||
|
||||
@@ -142,6 +142,15 @@ class TokenizerUtilsTest(unittest.TestCase):
|
||||
with self.subTest("Rust Tokenizer"):
|
||||
self.assertTrue(tokenizer_r("Small example to_encode").is_fast)
|
||||
|
||||
@require_tokenizers
|
||||
def test_batch_encoding_word_to_tokens(self):
|
||||
tokenizer_r = BertTokenizerFast.from_pretrained("bert-base-cased")
|
||||
encoded = tokenizer_r(["Test", "\xad", "test"], is_split_into_words=True)
|
||||
|
||||
self.assertEqual(encoded.word_to_tokens(0), TokenSpan(start=1, end=2))
|
||||
self.assertEqual(encoded.word_to_tokens(1), None)
|
||||
self.assertEqual(encoded.word_to_tokens(2), TokenSpan(start=2, end=3))
|
||||
|
||||
def test_batch_encoding_with_labels(self):
|
||||
batch = BatchEncoding({"inputs": [[1, 2, 3], [4, 5, 6]], "labels": [0, 1]})
|
||||
tensor_batch = batch.convert_to_tensors(tensor_type="np")
|
||||
|
||||
Reference in new issue
Block a user