Compare commits
20
Commits
v3.5.1hotfix
...
nlp
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fdcdd4ddc2 | ||
|
|
05c817a715 | ||
|
|
e6d732e649 | ||
|
|
b157d13a53 | ||
|
|
208d872719 | ||
|
|
2ffb7b58e3 | ||
|
|
8082dfc18b | ||
|
|
30a3972f6f | ||
|
|
04fe7318a0 | ||
|
|
81bc6da1aa | ||
|
|
625e334e28 | ||
|
|
57fb0c54a9 | ||
|
|
80f9f4bc00 | ||
|
|
beeee44743 | ||
|
|
de72cb314c | ||
|
|
db50c222a8 | ||
|
|
4d1647f1f1 | ||
|
|
dee993376d | ||
|
|
2b694f7358 | ||
|
|
e3896ba2d3 |
@@ -71,6 +71,18 @@ python run_glue.py \
|
||||
--learning_rate 2e-5 \
|
||||
--num_train_epochs 3.0 \
|
||||
--output_dir /tmp/$TASK_NAME/
|
||||
|
||||
python run_nlp_glue.py \
|
||||
--model_name_or_path bert-base-cased \
|
||||
--task_name mrpc \
|
||||
--do_train \
|
||||
--do_eval \
|
||||
--max_seq_length 128 \
|
||||
--per_device_train_batch_size 32 \
|
||||
--learning_rate 2e-5 \
|
||||
--num_train_epochs 3.0 \
|
||||
--output_dir /tmp/mrpc/ \
|
||||
--overwrite_output_dir
|
||||
```
|
||||
|
||||
where task name can be one of CoLA, SST-2, MRPC, STS-B, QQP, MNLI, QNLI, RTE, WNLI.
|
||||
|
||||
@@ -0,0 +1,289 @@
|
||||
# coding=utf-8
|
||||
# Copyright 2018 The Google AI Language Team Authors and The HuggingFace Inc. team.
|
||||
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
""" Finetuning the library models for sequence classification on GLUE (Bert, XLM, XLNet, RoBERTa, Albert, XLM-RoBERTa)."""
|
||||
|
||||
|
||||
import dataclasses
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Callable, Dict, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
import torch
|
||||
|
||||
import nlp
|
||||
|
||||
from transformers import AutoConfig, AutoModelForSequenceClassification, AutoTokenizer, EvalPrediction
|
||||
from transformers import (
|
||||
HfArgumentParser,
|
||||
Trainer,
|
||||
TrainingArguments,
|
||||
glue_compute_metrics,
|
||||
set_seed,
|
||||
)
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class GlueDataTrainingArguments:
|
||||
"""
|
||||
Arguments pertaining to what data we are going to input our model for training and eval.
|
||||
|
||||
Using `HfArgumentParser` we can turn this class
|
||||
into argparse arguments to be able to specify them on
|
||||
the command line.
|
||||
"""
|
||||
|
||||
task_name: str = field(metadata={"help": "The name of the task to train and/or evaluate on: "
|
||||
"['cola', 'sst2', 'mrpc', 'qqp', 'stsb', 'mnli', 'mnli_mismatched', 'mnli_matched', 'qnli', 'rte', 'wnli', 'ax']"})
|
||||
|
||||
max_seq_length: int = field(
|
||||
default=128,
|
||||
metadata={
|
||||
"help": "The maximum total input sequence length after tokenization. Sequences longer "
|
||||
"than this will be truncated, sequences shorter will be padded."
|
||||
},
|
||||
)
|
||||
|
||||
def __post_init__(self):
|
||||
self.task_name = self.task_name.lower().replace('-', '') # We used to allow 'sts-b' for 'stsb'
|
||||
|
||||
|
||||
@dataclass
|
||||
class ModelArguments:
|
||||
"""
|
||||
Arguments pertaining to which model/config/tokenizer we are going to fine-tune from.
|
||||
"""
|
||||
|
||||
model_name_or_path: str = field(
|
||||
metadata={"help": "Path to pretrained model or model identifier from huggingface.co/models"}
|
||||
)
|
||||
config_name: Optional[str] = field(
|
||||
default=None, metadata={"help": "Pretrained config name or path if not the same as model_name"}
|
||||
)
|
||||
tokenizer_name: Optional[str] = field(
|
||||
default=None, metadata={"help": "Pretrained tokenizer name or path if not the same as model_name"}
|
||||
)
|
||||
cache_dir: Optional[str] = field(
|
||||
default=None, metadata={"help": "Where do you want to store the pretrained models downloaded from s3"}
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
# See all possible arguments in src/transformers/training_args.py
|
||||
# or by passing the --help flag to this script.
|
||||
# We now keep distinct sets of args, for a cleaner separation of concerns.
|
||||
|
||||
parser = HfArgumentParser((ModelArguments, GlueDataTrainingArguments, TrainingArguments))
|
||||
|
||||
if len(sys.argv) == 2 and sys.argv[1].endswith(".json"):
|
||||
# If we pass only one argument to the script and it's the path to a json file,
|
||||
# let's parse it to get our arguments.
|
||||
model_args, data_args, training_args = parser.parse_json_file(json_file=os.path.abspath(sys.argv[1]))
|
||||
else:
|
||||
model_args, data_args, training_args = parser.parse_args_into_dataclasses()
|
||||
|
||||
if (
|
||||
os.path.exists(training_args.output_dir)
|
||||
and os.listdir(training_args.output_dir)
|
||||
and training_args.do_train
|
||||
and not training_args.overwrite_output_dir
|
||||
):
|
||||
raise ValueError(
|
||||
f"Output directory ({training_args.output_dir}) already exists and is not empty. Use --overwrite_output_dir to overcome."
|
||||
)
|
||||
|
||||
# Setup logging
|
||||
logging.basicConfig(
|
||||
format="%(asctime)s - %(levelname)s - %(name)s - %(message)s",
|
||||
datefmt="%m/%d/%Y %H:%M:%S",
|
||||
level=logging.INFO if training_args.local_rank in [-1, 0] else logging.WARN,
|
||||
)
|
||||
logger.warning(
|
||||
"Process rank: %s, device: %s, n_gpu: %s, distributed training: %s, 16-bits training: %s",
|
||||
training_args.local_rank,
|
||||
training_args.device,
|
||||
training_args.n_gpu,
|
||||
bool(training_args.local_rank != -1),
|
||||
training_args.fp16,
|
||||
)
|
||||
logger.info("Training/evaluation parameters %s", training_args)
|
||||
|
||||
# Set seed
|
||||
set_seed(training_args.seed)
|
||||
|
||||
# Get tokenizer
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
model_args.tokenizer_name if model_args.tokenizer_name else model_args.model_name_or_path,
|
||||
cache_dir=model_args.cache_dir,
|
||||
# use_fast=True,
|
||||
)
|
||||
|
||||
# Download, tokenize and prepare datasets for training a PyTorch model
|
||||
task_name = data_args.task_name.replace('-', '') # We used to allow 'sts-b' for 'stsb'
|
||||
glue = nlp.load_dataset('glue', name=task_name)
|
||||
|
||||
def tokenize(example):
|
||||
return tokenizer.batch_encode_plus(list(zip(example['sentence1'], example['sentence2'])),
|
||||
max_length=data_args.max_seq_length,
|
||||
pad_to_max_length=True)
|
||||
for split_name, split in glue.items():
|
||||
# Tokenize the dataset (this adds columns to the dataset)
|
||||
glue[split_name] = split.map(tokenize, batched=True)
|
||||
# Set the format of the dataset to output only the column our model can digest
|
||||
glue[split_name].set_format(columns=['input_ids', 'attention_mask', 'token_type_ids', 'label'])
|
||||
|
||||
# Get the splits
|
||||
train_dataset, eval_dataset, test_dataset = None, None, None
|
||||
if training_args.do_train:
|
||||
train_dataset = glue['train']
|
||||
if training_args.do_eval:
|
||||
eval_dataset = glue['validation' + ('_matched' if task_name == 'mnli' else '')]
|
||||
if training_args.do_predict:
|
||||
test_dataset = glue['test' + ('_matched' if task_name == 'mnli' else '')]
|
||||
|
||||
# Define output mode (regression or classification) and num labels (to define the size of the last layer of the model)
|
||||
output_mode = "regression" if 'float' in glue['train'].features['label'].dtype else "classification"
|
||||
if output_mode == "regression":
|
||||
num_labels = 1
|
||||
else:
|
||||
num_labels = len(train_dataset.unique('label'))
|
||||
|
||||
|
||||
if task_name in ["mnli", "mnli-mm"] and tokenizer.__class__.__name__ in (
|
||||
"RobertaTokenizer",
|
||||
"RobertaTokenizerFast",
|
||||
"XLMRobertaTokenizer",
|
||||
):
|
||||
raise NotImplementedError("""
|
||||
# HACK(label indices are swapped in RoBERTa pretrained model)
|
||||
label_list[1], label_list[2] = label_list[2], label_list[1]
|
||||
""")
|
||||
|
||||
# Load pretrained model and tokenizer
|
||||
#
|
||||
# Distributed training:
|
||||
# The .from_pretrained methods guarantee that only one local process can concurrently
|
||||
# download model & vocab.
|
||||
|
||||
config = AutoConfig.from_pretrained(
|
||||
model_args.config_name if model_args.config_name else model_args.model_name_or_path,
|
||||
num_labels=num_labels,
|
||||
finetuning_task=data_args.task_name,
|
||||
cache_dir=model_args.cache_dir,
|
||||
)
|
||||
model = AutoModelForSequenceClassification.from_pretrained(
|
||||
model_args.model_name_or_path,
|
||||
from_tf=bool(".ckpt" in model_args.model_name_or_path),
|
||||
config=config,
|
||||
cache_dir=model_args.cache_dir,
|
||||
)
|
||||
|
||||
def build_compute_metrics_fn(task_name: str) -> Callable[[EvalPrediction], Dict]:
|
||||
def compute_metrics_fn(p: EvalPrediction):
|
||||
if output_mode == "classification":
|
||||
preds = np.argmax(p.predictions, axis=1)
|
||||
elif output_mode == "regression":
|
||||
preds = np.squeeze(p.predictions)
|
||||
return glue_compute_metrics(task_name, preds, p.label_ids)
|
||||
|
||||
return compute_metrics_fn
|
||||
|
||||
# Initialize our Trainer
|
||||
trainer = Trainer(
|
||||
model=model,
|
||||
args=training_args,
|
||||
train_dataset=train_dataset,
|
||||
eval_dataset=eval_dataset,
|
||||
compute_metrics=build_compute_metrics_fn(data_args.task_name),
|
||||
)
|
||||
|
||||
# Training
|
||||
if training_args.do_train:
|
||||
trainer.train(
|
||||
model_path=model_args.model_name_or_path if os.path.isdir(model_args.model_name_or_path) else None
|
||||
)
|
||||
trainer.save_model()
|
||||
# For convenience, we also re-save the tokenizer to the same directory,
|
||||
# so that you can share your model easily on huggingface.co/models =)
|
||||
if trainer.is_world_master():
|
||||
tokenizer.save_pretrained(training_args.output_dir)
|
||||
|
||||
# Evaluation
|
||||
eval_results = {}
|
||||
if training_args.do_eval:
|
||||
logger.info("*** Evaluate ***")
|
||||
|
||||
# Loop to handle MNLI double evaluation (matched, mis-matched)
|
||||
eval_datasets = [eval_dataset]
|
||||
if data_args.task_name == "mnli":
|
||||
eval_datasets.append(glue["validation_mismatched"])
|
||||
|
||||
for eval_dataset in eval_datasets:
|
||||
trainer.compute_metrics = build_compute_metrics_fn(eval_dataset.args.task_name)
|
||||
eval_result = trainer.evaluate(eval_dataset=eval_dataset)
|
||||
|
||||
output_eval_file = os.path.join(
|
||||
training_args.output_dir, f"eval_results_{eval_dataset.args.task_name}.txt"
|
||||
)
|
||||
if trainer.is_world_master():
|
||||
with open(output_eval_file, "w") as writer:
|
||||
logger.info("***** Eval results {} *****".format(eval_dataset.args.task_name))
|
||||
for key, value in eval_result.items():
|
||||
logger.info(" %s = %s", key, value)
|
||||
writer.write("%s = %s\n" % (key, value))
|
||||
|
||||
eval_results.update(eval_result)
|
||||
|
||||
if training_args.do_predict:
|
||||
logging.info("*** Test ***")
|
||||
test_datasets = [test_dataset]
|
||||
if data_args.task_name == "mnli":
|
||||
test_datasets.append(glue["test_mismatched"])
|
||||
|
||||
for test_dataset in test_datasets:
|
||||
predictions = trainer.predict(test_dataset=test_dataset).predictions
|
||||
if output_mode == "classification":
|
||||
predictions = np.argmax(predictions, axis=1)
|
||||
|
||||
output_test_file = os.path.join(
|
||||
training_args.output_dir, f"test_results_{test_dataset.args.task_name}.txt"
|
||||
)
|
||||
if trainer.is_world_master():
|
||||
with open(output_test_file, "w") as writer:
|
||||
logger.info("***** Test results {} *****".format(test_dataset.args.task_name))
|
||||
writer.write("index\tprediction\n")
|
||||
for index, item in enumerate(predictions):
|
||||
if output_mode == "regression":
|
||||
writer.write("%d\t%3.3f\n" % (index, item))
|
||||
else:
|
||||
item = test_dataset.get_labels()[item]
|
||||
writer.write("%d\t%s\n" % (index, item))
|
||||
return eval_results
|
||||
|
||||
|
||||
def _mp_fn(index):
|
||||
# For xla_spawn (TPUs)
|
||||
main()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+170
-210
@@ -2,7 +2,7 @@
|
||||
# There's no way to ignore "F401 '...' imported but unused" warnings in this
|
||||
# module, but to preserve other warnings. So, don't check this module at all.
|
||||
|
||||
__version__ = "2.11.0"
|
||||
__version__ = "2.9.1"
|
||||
|
||||
# Work around to update TensorFlow's absl.logging threshold which alters the
|
||||
# default Python logging output behavior when present.
|
||||
@@ -19,6 +19,19 @@ else:
|
||||
|
||||
import logging
|
||||
|
||||
# Benchmarking
|
||||
from .benchmark_utils import (
|
||||
Frame,
|
||||
Memory,
|
||||
MemoryState,
|
||||
MemorySummary,
|
||||
MemoryTrace,
|
||||
UsedMemoryState,
|
||||
bytes_to_human_readable,
|
||||
start_memory_tracing,
|
||||
stop_memory_tracing,
|
||||
)
|
||||
|
||||
# Configurations
|
||||
from .configuration_albert import ALBERT_PRETRAINED_CONFIG_ARCHIVE_MAP, AlbertConfig
|
||||
from .configuration_auto import ALL_PRETRAINED_CONFIG_ARCHIVE_MAP, CONFIG_MAPPING, AutoConfig
|
||||
@@ -78,7 +91,6 @@ from .file_utils import (
|
||||
cached_path,
|
||||
is_tf_available,
|
||||
is_torch_available,
|
||||
is_torch_tpu_available,
|
||||
)
|
||||
from .hf_argparser import HfArgumentParser
|
||||
|
||||
@@ -118,7 +130,7 @@ from .pipelines import (
|
||||
# Tokenizers
|
||||
from .tokenization_albert import AlbertTokenizer
|
||||
from .tokenization_auto import TOKENIZER_MAPPING, AutoTokenizer
|
||||
from .tokenization_bart import BartTokenizer, BartTokenizerFast, MBartTokenizer
|
||||
from .tokenization_bart import BartTokenizer, MBartTokenizer
|
||||
from .tokenization_bert import BasicTokenizer, BertTokenizer, BertTokenizerFast, WordpieceTokenizer
|
||||
from .tokenization_bert_japanese import BertJapaneseTokenizer, CharacterTokenizer, MecabTokenizer
|
||||
from .tokenization_camembert import CamembertTokenizer
|
||||
@@ -127,27 +139,18 @@ from .tokenization_distilbert import DistilBertTokenizer, DistilBertTokenizerFas
|
||||
from .tokenization_electra import ElectraTokenizer, ElectraTokenizerFast
|
||||
from .tokenization_flaubert import FlaubertTokenizer
|
||||
from .tokenization_gpt2 import GPT2Tokenizer, GPT2TokenizerFast
|
||||
from .tokenization_longformer import LongformerTokenizer, LongformerTokenizerFast
|
||||
from .tokenization_longformer import LongformerTokenizer
|
||||
from .tokenization_openai import OpenAIGPTTokenizer, OpenAIGPTTokenizerFast
|
||||
from .tokenization_reformer import ReformerTokenizer
|
||||
from .tokenization_roberta import RobertaTokenizer, RobertaTokenizerFast
|
||||
from .tokenization_t5 import T5Tokenizer
|
||||
from .tokenization_transfo_xl import TransfoXLCorpus, TransfoXLTokenizer, TransfoXLTokenizerFast
|
||||
from .tokenization_utils import PreTrainedTokenizer
|
||||
from .tokenization_utils_base import (
|
||||
BatchEncoding,
|
||||
CharSpan,
|
||||
PreTrainedTokenizerBase,
|
||||
SpecialTokensMixin,
|
||||
TensorType,
|
||||
TokenSpan,
|
||||
)
|
||||
from .tokenization_utils_base import BatchEncoding, CharSpan, PreTrainedTokenizerBase, SpecialTokensMixin, TokenSpan
|
||||
from .tokenization_utils_fast import PreTrainedTokenizerFast
|
||||
from .tokenization_xlm import XLMTokenizer
|
||||
from .tokenization_xlm_roberta import XLMRobertaTokenizer
|
||||
from .tokenization_xlnet import SPIECE_UNDERLINE, XLNetTokenizer
|
||||
|
||||
# Trainer
|
||||
from .trainer_utils import EvalPrediction
|
||||
from .training_args import TrainingArguments
|
||||
from .training_args_tf import TFTrainingArguments
|
||||
@@ -169,17 +172,12 @@ if is_torch_available():
|
||||
AutoModelForSequenceClassification,
|
||||
AutoModelForQuestionAnswering,
|
||||
AutoModelWithLMHead,
|
||||
AutoModelForCausalLM,
|
||||
AutoModelForMaskedLM,
|
||||
AutoModelForSeq2SeqLM,
|
||||
AutoModelForTokenClassification,
|
||||
AutoModelForMultipleChoice,
|
||||
ALL_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
MODEL_MAPPING,
|
||||
MODEL_FOR_PRETRAINING_MAPPING,
|
||||
MODEL_WITH_LM_HEAD_MAPPING,
|
||||
MODEL_FOR_CAUSAL_LM_MAPPING,
|
||||
MODEL_FOR_MASKED_LM_MAPPING,
|
||||
MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING,
|
||||
MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING,
|
||||
MODEL_FOR_QUESTION_ANSWERING_MAPPING,
|
||||
MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING,
|
||||
@@ -191,14 +189,13 @@ if is_torch_available():
|
||||
BertModel,
|
||||
BertForPreTraining,
|
||||
BertForMaskedLM,
|
||||
BertLMHeadModel,
|
||||
BertForNextSentencePrediction,
|
||||
BertForSequenceClassification,
|
||||
BertForMultipleChoice,
|
||||
BertForTokenClassification,
|
||||
BertForQuestionAnswering,
|
||||
load_tf_weights_in_bert,
|
||||
BERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
BERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
BertLayer,
|
||||
)
|
||||
from .modeling_openai import (
|
||||
@@ -207,7 +204,7 @@ if is_torch_available():
|
||||
OpenAIGPTLMHeadModel,
|
||||
OpenAIGPTDoubleHeadsModel,
|
||||
load_tf_weights_in_openai_gpt,
|
||||
OPENAI_GPT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
OPENAI_GPT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_transfo_xl import (
|
||||
TransfoXLPreTrainedModel,
|
||||
@@ -215,7 +212,7 @@ if is_torch_available():
|
||||
TransfoXLLMHeadModel,
|
||||
AdaptiveEmbedding,
|
||||
load_tf_weights_in_transfo_xl,
|
||||
TRANSFO_XL_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TRANSFO_XL_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_gpt2 import (
|
||||
GPT2PreTrainedModel,
|
||||
@@ -223,9 +220,9 @@ if is_torch_available():
|
||||
GPT2LMHeadModel,
|
||||
GPT2DoubleHeadsModel,
|
||||
load_tf_weights_in_gpt2,
|
||||
GPT2_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
GPT2_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_ctrl import CTRLPreTrainedModel, CTRLModel, CTRLLMHeadModel, CTRL_PRETRAINED_MODEL_ARCHIVE_LIST
|
||||
from .modeling_ctrl import CTRLPreTrainedModel, CTRLModel, CTRLLMHeadModel, CTRL_PRETRAINED_MODEL_ARCHIVE_MAP
|
||||
from .modeling_xlnet import (
|
||||
XLNetPreTrainedModel,
|
||||
XLNetModel,
|
||||
@@ -236,7 +233,7 @@ if is_torch_available():
|
||||
XLNetForQuestionAnsweringSimple,
|
||||
XLNetForQuestionAnswering,
|
||||
load_tf_weights_in_xlnet,
|
||||
XLNET_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
XLNET_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_xlm import (
|
||||
XLMPreTrainedModel,
|
||||
@@ -246,15 +243,13 @@ if is_torch_available():
|
||||
XLMForTokenClassification,
|
||||
XLMForQuestionAnswering,
|
||||
XLMForQuestionAnsweringSimple,
|
||||
XLM_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
XLM_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_bart import (
|
||||
PretrainedBartModel,
|
||||
BartForSequenceClassification,
|
||||
BartModel,
|
||||
BartForConditionalGeneration,
|
||||
BartForQuestionAnswering,
|
||||
BART_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
BART_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_marian import MarianMTModel
|
||||
from .tokenization_marian import MarianTokenizer
|
||||
@@ -265,17 +260,16 @@ if is_torch_available():
|
||||
RobertaForMultipleChoice,
|
||||
RobertaForTokenClassification,
|
||||
RobertaForQuestionAnswering,
|
||||
ROBERTA_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
ROBERTA_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_distilbert import (
|
||||
DistilBertPreTrainedModel,
|
||||
DistilBertForMaskedLM,
|
||||
DistilBertModel,
|
||||
DistilBertForMultipleChoice,
|
||||
DistilBertForSequenceClassification,
|
||||
DistilBertForQuestionAnswering,
|
||||
DistilBertForTokenClassification,
|
||||
DISTILBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
DISTILBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_camembert import (
|
||||
CamembertForMaskedLM,
|
||||
@@ -284,7 +278,7 @@ if is_torch_available():
|
||||
CamembertForMultipleChoice,
|
||||
CamembertForTokenClassification,
|
||||
CamembertForQuestionAnswering,
|
||||
CAMEMBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
CAMEMBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_encoder_decoder import EncoderDecoderModel
|
||||
from .modeling_t5 import (
|
||||
@@ -292,19 +286,18 @@ if is_torch_available():
|
||||
T5Model,
|
||||
T5ForConditionalGeneration,
|
||||
load_tf_weights_in_t5,
|
||||
T5_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
T5_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_albert import (
|
||||
AlbertPreTrainedModel,
|
||||
AlbertModel,
|
||||
AlbertForPreTraining,
|
||||
AlbertForMaskedLM,
|
||||
AlbertForMultipleChoice,
|
||||
AlbertForSequenceClassification,
|
||||
AlbertForQuestionAnswering,
|
||||
AlbertForTokenClassification,
|
||||
load_tf_weights_in_albert,
|
||||
ALBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
ALBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_xlm_roberta import (
|
||||
XLMRobertaForMaskedLM,
|
||||
@@ -312,8 +305,7 @@ if is_torch_available():
|
||||
XLMRobertaForMultipleChoice,
|
||||
XLMRobertaForSequenceClassification,
|
||||
XLMRobertaForTokenClassification,
|
||||
XLMRobertaForQuestionAnswering,
|
||||
XLM_ROBERTA_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
XLM_ROBERTA_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
from .modeling_mmbt import ModalEmbeddings, MMBTModel, MMBTForClassification
|
||||
|
||||
@@ -323,7 +315,7 @@ if is_torch_available():
|
||||
FlaubertForSequenceClassification,
|
||||
FlaubertForQuestionAnswering,
|
||||
FlaubertForQuestionAnsweringSimple,
|
||||
FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
FLAUBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_electra import (
|
||||
@@ -331,11 +323,9 @@ if is_torch_available():
|
||||
ElectraForMaskedLM,
|
||||
ElectraForTokenClassification,
|
||||
ElectraPreTrainedModel,
|
||||
ElectraForSequenceClassification,
|
||||
ElectraForQuestionAnswering,
|
||||
ElectraModel,
|
||||
load_tf_weights_in_electra,
|
||||
ELECTRA_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
ELECTRA_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_reformer import (
|
||||
@@ -343,18 +333,10 @@ if is_torch_available():
|
||||
ReformerLayer,
|
||||
ReformerModel,
|
||||
ReformerModelWithLMHead,
|
||||
REFORMER_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
REFORMER_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_longformer import (
|
||||
LongformerModel,
|
||||
LongformerForMaskedLM,
|
||||
LongformerForSequenceClassification,
|
||||
LongformerForMultipleChoice,
|
||||
LongformerForTokenClassification,
|
||||
LongformerForQuestionAnswering,
|
||||
LONGFORMER_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
)
|
||||
from .modeling_longformer import LONGFORMER_PRETRAINED_MODEL_ARCHIVE_MAP, LongformerModel, LongformerForMaskedLM
|
||||
|
||||
# Optimization
|
||||
from .optimization import (
|
||||
@@ -368,202 +350,180 @@ if is_torch_available():
|
||||
|
||||
# Trainer
|
||||
from .trainer import Trainer, set_seed, torch_distributed_zero_first, EvalPrediction
|
||||
from .data.data_collator import default_data_collator, DataCollator, DataCollatorForLanguageModeling
|
||||
from .data.data_collator import DefaultDataCollator, DataCollator, DataCollatorForLanguageModeling
|
||||
from .data.datasets import GlueDataset, TextDataset, LineByLineTextDataset, GlueDataTrainingArguments
|
||||
|
||||
# Benchmarks
|
||||
from .benchmark import PyTorchBenchmark, PyTorchBenchmarkArguments
|
||||
|
||||
# TensorFlow
|
||||
if is_tf_available():
|
||||
from .modeling_tf_utils import (
|
||||
TFPreTrainedModel,
|
||||
TFSharedEmbeddings,
|
||||
TFSequenceSummary,
|
||||
shape_list,
|
||||
tf_top_k_top_p_filtering,
|
||||
TFPreTrainedModel,
|
||||
TFSequenceSummary,
|
||||
TFSharedEmbeddings,
|
||||
)
|
||||
from .modeling_tf_auto import (
|
||||
TF_MODEL_MAPPING,
|
||||
TF_MODEL_FOR_MULTIPLE_CHOICE_MAPPING,
|
||||
TF_MODEL_FOR_PRETRAINING_MAPPING,
|
||||
TF_MODEL_FOR_QUESTION_ANSWERING_MAPPING,
|
||||
TF_MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING,
|
||||
TF_MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING,
|
||||
TF_MODEL_WITH_LM_HEAD_MAPPING,
|
||||
TFAutoModel,
|
||||
TFAutoModelForMultipleChoice,
|
||||
TFAutoModelForPreTraining,
|
||||
TFAutoModelForQuestionAnswering,
|
||||
TFAutoModelForMultipleChoice,
|
||||
TFAutoModelForSequenceClassification,
|
||||
TFAutoModelForTokenClassification,
|
||||
TFAutoModelForQuestionAnswering,
|
||||
TFAutoModelWithLMHead,
|
||||
)
|
||||
|
||||
from .modeling_tf_albert import (
|
||||
TF_ALBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFAlbertForMaskedLM,
|
||||
TFAlbertForMultipleChoice,
|
||||
TFAlbertForPreTraining,
|
||||
TFAlbertForQuestionAnswering,
|
||||
TFAlbertForSequenceClassification,
|
||||
TFAlbertForTokenClassification,
|
||||
TFAlbertMainLayer,
|
||||
TFAlbertModel,
|
||||
TFAlbertPreTrainedModel,
|
||||
TFAutoModelForTokenClassification,
|
||||
TF_ALL_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
TF_MODEL_MAPPING,
|
||||
TF_MODEL_FOR_PRETRAINING_MAPPING,
|
||||
TF_MODEL_WITH_LM_HEAD_MAPPING,
|
||||
TF_MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING,
|
||||
TF_MODEL_FOR_QUESTION_ANSWERING_MAPPING,
|
||||
TF_MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING,
|
||||
)
|
||||
|
||||
from .modeling_tf_bert import (
|
||||
TF_BERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFBertEmbeddings,
|
||||
TFBertForMaskedLM,
|
||||
TFBertForMultipleChoice,
|
||||
TFBertForNextSentencePrediction,
|
||||
TFBertForPreTraining,
|
||||
TFBertForQuestionAnswering,
|
||||
TFBertForSequenceClassification,
|
||||
TFBertForTokenClassification,
|
||||
TFBertMainLayer,
|
||||
TFBertModel,
|
||||
TFBertPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_camembert import (
|
||||
TF_CAMEMBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFCamembertForMaskedLM,
|
||||
TFCamembertModel,
|
||||
TFCamembertForMultipleChoice,
|
||||
TFCamembertForQuestionAnswering,
|
||||
TFCamembertForSequenceClassification,
|
||||
TFCamembertForTokenClassification,
|
||||
)
|
||||
|
||||
from .modeling_tf_ctrl import (
|
||||
TF_CTRL_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFCTRLLMHeadModel,
|
||||
TFCTRLModel,
|
||||
TFCTRLPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_distilbert import (
|
||||
TF_DISTILBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFDistilBertForMaskedLM,
|
||||
TFDistilBertForMultipleChoice,
|
||||
TFDistilBertForQuestionAnswering,
|
||||
TFDistilBertForSequenceClassification,
|
||||
TFDistilBertForTokenClassification,
|
||||
TFDistilBertMainLayer,
|
||||
TFDistilBertModel,
|
||||
TFDistilBertPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_electra import (
|
||||
TF_ELECTRA_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFElectraForMaskedLM,
|
||||
TFElectraForPreTraining,
|
||||
TFElectraForQuestionAnswering,
|
||||
TFElectraForTokenClassification,
|
||||
TFElectraModel,
|
||||
TFElectraPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_flaubert import (
|
||||
TF_FLAUBERT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFFlaubertForMultipleChoice,
|
||||
TFFlaubertForQuestionAnsweringSimple,
|
||||
TFFlaubertForSequenceClassification,
|
||||
TFFlaubertForTokenClassification,
|
||||
TFFlaubertWithLMHeadModel,
|
||||
TFFlaubertModel,
|
||||
TFBertMainLayer,
|
||||
TFBertEmbeddings,
|
||||
TFBertModel,
|
||||
TFBertForPreTraining,
|
||||
TFBertForMaskedLM,
|
||||
TFBertForNextSentencePrediction,
|
||||
TFBertForSequenceClassification,
|
||||
TFBertForMultipleChoice,
|
||||
TFBertForTokenClassification,
|
||||
TFBertForQuestionAnswering,
|
||||
TF_BERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_gpt2 import (
|
||||
TF_GPT2_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFGPT2DoubleHeadsModel,
|
||||
TFGPT2LMHeadModel,
|
||||
TFGPT2PreTrainedModel,
|
||||
TFGPT2MainLayer,
|
||||
TFGPT2Model,
|
||||
TFGPT2PreTrainedModel,
|
||||
TFGPT2LMHeadModel,
|
||||
TFGPT2DoubleHeadsModel,
|
||||
TF_GPT2_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_openai import (
|
||||
TF_OPENAI_GPT_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFOpenAIGPTDoubleHeadsModel,
|
||||
TFOpenAIGPTLMHeadModel,
|
||||
TFOpenAIGPTPreTrainedModel,
|
||||
TFOpenAIGPTMainLayer,
|
||||
TFOpenAIGPTModel,
|
||||
TFOpenAIGPTPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_roberta import (
|
||||
TF_ROBERTA_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFRobertaForMaskedLM,
|
||||
TFRobertaForMultipleChoice,
|
||||
TFRobertaForQuestionAnswering,
|
||||
TFRobertaForSequenceClassification,
|
||||
TFRobertaForTokenClassification,
|
||||
TFRobertaMainLayer,
|
||||
TFRobertaModel,
|
||||
TFRobertaPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_t5 import (
|
||||
TF_T5_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFT5ForConditionalGeneration,
|
||||
TFT5Model,
|
||||
TFT5PreTrainedModel,
|
||||
TFOpenAIGPTLMHeadModel,
|
||||
TFOpenAIGPTDoubleHeadsModel,
|
||||
TF_OPENAI_GPT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_transfo_xl import (
|
||||
TF_TRANSFO_XL_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFAdaptiveEmbedding,
|
||||
TFTransfoXLLMHeadModel,
|
||||
TFTransfoXLPreTrainedModel,
|
||||
TFTransfoXLMainLayer,
|
||||
TFTransfoXLModel,
|
||||
TFTransfoXLPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_xlm import (
|
||||
TF_XLM_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFXLMForMultipleChoice,
|
||||
TFXLMForQuestionAnsweringSimple,
|
||||
TFXLMForSequenceClassification,
|
||||
TFXLMForTokenClassification,
|
||||
TFXLMWithLMHeadModel,
|
||||
TFXLMMainLayer,
|
||||
TFXLMModel,
|
||||
TFXLMPreTrainedModel,
|
||||
)
|
||||
|
||||
from .modeling_tf_xlm_roberta import (
|
||||
TF_XLM_ROBERTA_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFXLMRobertaForMaskedLM,
|
||||
TFXLMRobertaForMultipleChoice,
|
||||
TFXLMRobertaForQuestionAnswering,
|
||||
TFXLMRobertaForSequenceClassification,
|
||||
TFXLMRobertaForTokenClassification,
|
||||
TFXLMRobertaModel,
|
||||
TFTransfoXLLMHeadModel,
|
||||
TF_TRANSFO_XL_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
TFAdaptiveEmbedding,
|
||||
)
|
||||
|
||||
from .modeling_tf_xlnet import (
|
||||
TF_XLNET_PRETRAINED_MODEL_ARCHIVE_LIST,
|
||||
TFXLNetForMultipleChoice,
|
||||
TFXLNetForQuestionAnsweringSimple,
|
||||
TFXLNetForSequenceClassification,
|
||||
TFXLNetForTokenClassification,
|
||||
TFXLNetLMHeadModel,
|
||||
TFXLNetPreTrainedModel,
|
||||
TFXLNetMainLayer,
|
||||
TFXLNetModel,
|
||||
TFXLNetPreTrainedModel,
|
||||
TFXLNetLMHeadModel,
|
||||
TFXLNetForSequenceClassification,
|
||||
TFXLNetForTokenClassification,
|
||||
TFXLNetForQuestionAnsweringSimple,
|
||||
TF_XLNET_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_xlm import (
|
||||
TFXLMPreTrainedModel,
|
||||
TFXLMMainLayer,
|
||||
TFXLMModel,
|
||||
TFXLMWithLMHeadModel,
|
||||
TFXLMForSequenceClassification,
|
||||
TFXLMForQuestionAnsweringSimple,
|
||||
TF_XLM_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_xlm_roberta import (
|
||||
TFXLMRobertaForMaskedLM,
|
||||
TFXLMRobertaModel,
|
||||
TFXLMRobertaForSequenceClassification,
|
||||
TFXLMRobertaForTokenClassification,
|
||||
TF_XLM_ROBERTA_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_roberta import (
|
||||
TFRobertaPreTrainedModel,
|
||||
TFRobertaMainLayer,
|
||||
TFRobertaModel,
|
||||
TFRobertaForMaskedLM,
|
||||
TFRobertaForSequenceClassification,
|
||||
TFRobertaForTokenClassification,
|
||||
TFRobertaForQuestionAnswering,
|
||||
TF_ROBERTA_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_camembert import (
|
||||
TFCamembertModel,
|
||||
TFCamembertForMaskedLM,
|
||||
TFCamembertForSequenceClassification,
|
||||
TFCamembertForTokenClassification,
|
||||
TF_CAMEMBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_flaubert import (
|
||||
TFFlaubertModel,
|
||||
TFFlaubertWithLMHeadModel,
|
||||
TFFlaubertForSequenceClassification,
|
||||
TF_FLAUBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_distilbert import (
|
||||
TFDistilBertPreTrainedModel,
|
||||
TFDistilBertMainLayer,
|
||||
TFDistilBertModel,
|
||||
TFDistilBertForMaskedLM,
|
||||
TFDistilBertForSequenceClassification,
|
||||
TFDistilBertForTokenClassification,
|
||||
TFDistilBertForQuestionAnswering,
|
||||
TF_DISTILBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_ctrl import (
|
||||
TFCTRLPreTrainedModel,
|
||||
TFCTRLModel,
|
||||
TFCTRLLMHeadModel,
|
||||
TF_CTRL_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_albert import (
|
||||
TFAlbertPreTrainedModel,
|
||||
TFAlbertMainLayer,
|
||||
TFAlbertModel,
|
||||
TFAlbertForPreTraining,
|
||||
TFAlbertForMaskedLM,
|
||||
TFAlbertForMultipleChoice,
|
||||
TFAlbertForSequenceClassification,
|
||||
TFAlbertForQuestionAnswering,
|
||||
TF_ALBERT_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_t5 import (
|
||||
TFT5PreTrainedModel,
|
||||
TFT5Model,
|
||||
TFT5ForConditionalGeneration,
|
||||
TF_T5_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
from .modeling_tf_electra import (
|
||||
TFElectraPreTrainedModel,
|
||||
TFElectraModel,
|
||||
TFElectraForPreTraining,
|
||||
TFElectraForMaskedLM,
|
||||
TFElectraForTokenClassification,
|
||||
TF_ELECTRA_PRETRAINED_MODEL_ARCHIVE_MAP,
|
||||
)
|
||||
|
||||
# Optimization
|
||||
from .optimization_tf import (
|
||||
AdamWeightDecay,
|
||||
create_optimizer,
|
||||
GradientAccumulator,
|
||||
WarmUp,
|
||||
)
|
||||
from .optimization_tf import WarmUp, create_optimizer, AdamWeightDecay, GradientAccumulator
|
||||
|
||||
# Trainer
|
||||
from .trainer_tf import TFTrainer
|
||||
|
||||
@@ -23,33 +23,22 @@ import tensorflow as tf
|
||||
from .configuration_albert import AlbertConfig
|
||||
from .file_utils import MULTIPLE_CHOICE_DUMMY_INPUTS, add_start_docstrings, add_start_docstrings_to_callable
|
||||
from .modeling_tf_bert import ACT2FN, TFBertSelfAttention
|
||||
from .modeling_tf_utils import (
|
||||
TFMultipleChoiceLoss,
|
||||
TFPreTrainedModel,
|
||||
TFQuestionAnsweringLoss,
|
||||
TFSequenceClassificationLoss,
|
||||
TFTokenClassificationLoss,
|
||||
cast_bool_to_primitive,
|
||||
get_initializer,
|
||||
keras_serializable,
|
||||
shape_list,
|
||||
)
|
||||
from .modeling_tf_utils import TFPreTrainedModel, get_initializer, keras_serializable, shape_list
|
||||
from .tokenization_utils import BatchEncoding
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
TF_ALBERT_PRETRAINED_MODEL_ARCHIVE_LIST = [
|
||||
"albert-base-v1",
|
||||
"albert-large-v1",
|
||||
"albert-xlarge-v1",
|
||||
"albert-xxlarge-v1",
|
||||
"albert-base-v2",
|
||||
"albert-large-v2",
|
||||
"albert-xlarge-v2",
|
||||
"albert-xxlarge-v2",
|
||||
# See all ALBERT models at https://huggingface.co/models?filter=albert
|
||||
]
|
||||
TF_ALBERT_PRETRAINED_MODEL_ARCHIVE_MAP = {
|
||||
"albert-base-v1": "https://cdn.huggingface.co/albert-base-v1-with-prefix-tf_model.h5",
|
||||
"albert-large-v1": "https://cdn.huggingface.co/albert-large-v1-with-prefix-tf_model.h5",
|
||||
"albert-xlarge-v1": "https://cdn.huggingface.co/albert-xlarge-v1-with-prefix-tf_model.h5",
|
||||
"albert-xxlarge-v1": "https://cdn.huggingface.co/albert-xxlarge-v1-with-prefix-tf_model.h5",
|
||||
"albert-base-v2": "https://cdn.huggingface.co/albert-base-v2-with-prefix-tf_model.h5",
|
||||
"albert-large-v2": "https://cdn.huggingface.co/albert-large-v2-with-prefix-tf_model.h5",
|
||||
"albert-xlarge-v2": "https://cdn.huggingface.co/albert-xlarge-v2-with-prefix-tf_model.h5",
|
||||
"albert-xxlarge-v2": "https://cdn.huggingface.co/albert-xxlarge-v2-with-prefix-tf_model.h5",
|
||||
}
|
||||
|
||||
|
||||
class TFAlbertEmbeddings(tf.keras.layers.Layer):
|
||||
@@ -159,6 +148,7 @@ class TFAlbertSelfAttention(tf.keras.layers.Layer):
|
||||
"The hidden size (%d) is not a multiple of the number of attention "
|
||||
"heads (%d)" % (config.hidden_size, config.num_attention_heads)
|
||||
)
|
||||
self.output_attentions = config.output_attentions
|
||||
|
||||
self.num_attention_heads = config.num_attention_heads
|
||||
assert config.hidden_size % config.num_attention_heads == 0
|
||||
@@ -182,7 +172,7 @@ class TFAlbertSelfAttention(tf.keras.layers.Layer):
|
||||
return tf.transpose(x, perm=[0, 2, 1, 3])
|
||||
|
||||
def call(self, inputs, training=False):
|
||||
hidden_states, attention_mask, head_mask, output_attentions = inputs
|
||||
hidden_states, attention_mask, head_mask = inputs
|
||||
|
||||
batch_size = shape_list(hidden_states)[0]
|
||||
mixed_query_layer = self.query(hidden_states)
|
||||
@@ -222,9 +212,7 @@ class TFAlbertSelfAttention(tf.keras.layers.Layer):
|
||||
context_layer, (batch_size, -1, self.all_head_size)
|
||||
) # (batch_size, seq_len_q, all_head_size)
|
||||
|
||||
outputs = (
|
||||
(context_layer, attention_probs) if cast_bool_to_primitive(output_attentions) is True else (context_layer,)
|
||||
)
|
||||
outputs = (context_layer, attention_probs) if self.output_attentions else (context_layer,)
|
||||
return outputs
|
||||
|
||||
|
||||
@@ -261,7 +249,7 @@ class TFAlbertAttention(TFBertSelfAttention):
|
||||
raise NotImplementedError
|
||||
|
||||
def call(self, inputs, training=False):
|
||||
input_tensor, attention_mask, head_mask, output_attentions = inputs
|
||||
input_tensor, attention_mask, head_mask = inputs
|
||||
|
||||
batch_size = shape_list(input_tensor)[0]
|
||||
mixed_query_layer = self.query(input_tensor)
|
||||
@@ -301,9 +289,7 @@ class TFAlbertAttention(TFBertSelfAttention):
|
||||
context_layer, (batch_size, -1, self.all_head_size)
|
||||
) # (batch_size, seq_len_q, all_head_size)
|
||||
|
||||
self_outputs = (
|
||||
(context_layer, attention_probs) if cast_bool_to_primitive(output_attentions) is True else (context_layer,)
|
||||
)
|
||||
self_outputs = (context_layer, attention_probs) if self.output_attentions else (context_layer,)
|
||||
|
||||
hidden_states = self_outputs[0]
|
||||
|
||||
@@ -339,11 +325,9 @@ class TFAlbertLayer(tf.keras.layers.Layer):
|
||||
self.dropout = tf.keras.layers.Dropout(config.hidden_dropout_prob)
|
||||
|
||||
def call(self, inputs, training=False):
|
||||
hidden_states, attention_mask, head_mask, output_attentions = inputs
|
||||
hidden_states, attention_mask, head_mask = inputs
|
||||
|
||||
attention_outputs = self.attention(
|
||||
[hidden_states, attention_mask, head_mask, output_attentions], training=training
|
||||
)
|
||||
attention_outputs = self.attention([hidden_states, attention_mask, head_mask], training=training)
|
||||
ffn_output = self.ffn(attention_outputs[0])
|
||||
ffn_output = self.activation(ffn_output)
|
||||
ffn_output = self.ffn_output(ffn_output)
|
||||
@@ -360,24 +344,23 @@ class TFAlbertLayerGroup(tf.keras.layers.Layer):
|
||||
def __init__(self, config, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
self.output_attentions = config.output_attentions
|
||||
self.output_hidden_states = config.output_hidden_states
|
||||
self.albert_layers = [
|
||||
TFAlbertLayer(config, name="albert_layers_._{}".format(i)) for i in range(config.inner_group_num)
|
||||
]
|
||||
|
||||
def call(self, inputs, training=False):
|
||||
hidden_states, attention_mask, head_mask, output_attentions = inputs
|
||||
hidden_states, attention_mask, head_mask = inputs
|
||||
|
||||
layer_hidden_states = ()
|
||||
layer_attentions = ()
|
||||
|
||||
for layer_index, albert_layer in enumerate(self.albert_layers):
|
||||
layer_output = albert_layer(
|
||||
[hidden_states, attention_mask, head_mask[layer_index], output_attentions], training=training
|
||||
)
|
||||
layer_output = albert_layer([hidden_states, attention_mask, head_mask[layer_index]], training=training)
|
||||
hidden_states = layer_output[0]
|
||||
|
||||
if cast_bool_to_primitive(output_attentions) is True:
|
||||
if self.output_attentions:
|
||||
layer_attentions = layer_attentions + (layer_output[1],)
|
||||
|
||||
if self.output_hidden_states:
|
||||
@@ -386,7 +369,7 @@ class TFAlbertLayerGroup(tf.keras.layers.Layer):
|
||||
outputs = (hidden_states,)
|
||||
if self.output_hidden_states:
|
||||
outputs = outputs + (layer_hidden_states,)
|
||||
if cast_bool_to_primitive(output_attentions) is True:
|
||||
if self.output_attentions:
|
||||
outputs = outputs + (layer_attentions,)
|
||||
# last-layer hidden state, (layer hidden states), (layer attentions)
|
||||
return outputs
|
||||
@@ -397,6 +380,7 @@ class TFAlbertTransformer(tf.keras.layers.Layer):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
self.config = config
|
||||
self.output_attentions = config.output_attentions
|
||||
self.output_hidden_states = config.output_hidden_states
|
||||
self.embedding_hidden_mapping_in = tf.keras.layers.Dense(
|
||||
config.hidden_size,
|
||||
@@ -409,7 +393,7 @@ class TFAlbertTransformer(tf.keras.layers.Layer):
|
||||
]
|
||||
|
||||
def call(self, inputs, training=False):
|
||||
hidden_states, attention_mask, head_mask, output_attentions = inputs
|
||||
hidden_states, attention_mask, head_mask = inputs
|
||||
|
||||
hidden_states = self.embedding_hidden_mapping_in(hidden_states)
|
||||
all_attentions = ()
|
||||
@@ -429,13 +413,12 @@ class TFAlbertTransformer(tf.keras.layers.Layer):
|
||||
hidden_states,
|
||||
attention_mask,
|
||||
head_mask[group_idx * layers_per_group : (group_idx + 1) * layers_per_group],
|
||||
output_attentions,
|
||||
],
|
||||
training=training,
|
||||
)
|
||||
hidden_states = layer_group_output[0]
|
||||
|
||||
if cast_bool_to_primitive(output_attentions) is True:
|
||||
if self.output_attentions:
|
||||
all_attentions = all_attentions + layer_group_output[-1]
|
||||
|
||||
if self.output_hidden_states:
|
||||
@@ -444,7 +427,7 @@ class TFAlbertTransformer(tf.keras.layers.Layer):
|
||||
outputs = (hidden_states,)
|
||||
if self.output_hidden_states:
|
||||
outputs = outputs + (all_hidden_states,)
|
||||
if cast_bool_to_primitive(output_attentions) is True:
|
||||
if self.output_attentions:
|
||||
outputs = outputs + (all_attentions,)
|
||||
|
||||
# last-layer hidden state, (all hidden states), (all attentions)
|
||||
@@ -457,6 +440,7 @@ class TFAlbertPreTrainedModel(TFPreTrainedModel):
|
||||
"""
|
||||
|
||||
config_class = AlbertConfig
|
||||
pretrained_model_archive_map = TF_ALBERT_PRETRAINED_MODEL_ARCHIVE_MAP
|
||||
base_model_prefix = "albert"
|
||||
|
||||
|
||||
@@ -501,7 +485,6 @@ class TFAlbertMainLayer(tf.keras.layers.Layer):
|
||||
def __init__(self, config, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.num_hidden_layers = config.num_hidden_layers
|
||||
self.output_attentions = config.output_attentions
|
||||
|
||||
self.embeddings = TFAlbertEmbeddings(config, name="embeddings")
|
||||
self.encoder = TFAlbertTransformer(config, name="encoder")
|
||||
@@ -533,7 +516,6 @@ class TFAlbertMainLayer(tf.keras.layers.Layer):
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
inputs_embeds=None,
|
||||
output_attentions=None,
|
||||
training=False,
|
||||
):
|
||||
if isinstance(inputs, (tuple, list)):
|
||||
@@ -543,8 +525,7 @@ class TFAlbertMainLayer(tf.keras.layers.Layer):
|
||||
position_ids = inputs[3] if len(inputs) > 3 else position_ids
|
||||
head_mask = inputs[4] if len(inputs) > 4 else head_mask
|
||||
inputs_embeds = inputs[5] if len(inputs) > 5 else inputs_embeds
|
||||
output_attentions = inputs[6] if len(inputs) > 6 else output_attentions
|
||||
assert len(inputs) <= 7, "Too many inputs."
|
||||
assert len(inputs) <= 6, "Too many inputs."
|
||||
elif isinstance(inputs, (dict, BatchEncoding)):
|
||||
input_ids = inputs.get("input_ids")
|
||||
attention_mask = inputs.get("attention_mask", attention_mask)
|
||||
@@ -552,13 +533,10 @@ class TFAlbertMainLayer(tf.keras.layers.Layer):
|
||||
position_ids = inputs.get("position_ids", position_ids)
|
||||
head_mask = inputs.get("head_mask", head_mask)
|
||||
inputs_embeds = inputs.get("inputs_embeds", inputs_embeds)
|
||||
output_attentions = inputs.get("output_attentions", output_attentions)
|
||||
assert len(inputs) <= 7, "Too many inputs."
|
||||
assert len(inputs) <= 6, "Too many inputs."
|
||||
else:
|
||||
input_ids = inputs
|
||||
|
||||
output_attentions = output_attentions if output_attentions is not None else self.output_attentions
|
||||
|
||||
if input_ids is not None and inputs_embeds is not None:
|
||||
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
||||
elif input_ids is not None:
|
||||
@@ -601,9 +579,7 @@ class TFAlbertMainLayer(tf.keras.layers.Layer):
|
||||
# head_mask = tf.constant([0] * self.num_hidden_layers)
|
||||
|
||||
embedding_output = self.embeddings([input_ids, position_ids, token_type_ids, inputs_embeds], training=training)
|
||||
encoder_outputs = self.encoder(
|
||||
[embedding_output, extended_attention_mask, head_mask, output_attentions], training=training
|
||||
)
|
||||
encoder_outputs = self.encoder([embedding_output, extended_attention_mask, head_mask], training=training)
|
||||
|
||||
sequence_output = encoder_outputs[0]
|
||||
pooled_output = self.pooler(sequence_output[:, 0])
|
||||
@@ -652,7 +628,7 @@ ALBERT_START_DOCSTRING = r"""
|
||||
|
||||
ALBERT_INPUTS_DOCSTRING = r"""
|
||||
Args:
|
||||
input_ids (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`{0}`):
|
||||
input_ids (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length)`):
|
||||
Indices of input sequence tokens in the vocabulary.
|
||||
|
||||
Indices can be obtained using :class:`transformers.AlbertTokenizer`.
|
||||
@@ -660,19 +636,19 @@ ALBERT_INPUTS_DOCSTRING = r"""
|
||||
:func:`transformers.PreTrainedTokenizer.encode_plus` for details.
|
||||
|
||||
`What are input IDs? <../glossary.html#input-ids>`__
|
||||
attention_mask (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`{0}`, `optional, defaults to :obj:`None`):
|
||||
attention_mask (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length)`, `optional, defaults to :obj:`None`):
|
||||
Mask to avoid performing attention on padding token indices.
|
||||
Mask values selected in ``[0, 1]``:
|
||||
``1`` for tokens that are NOT MASKED, ``0`` for MASKED tokens.
|
||||
|
||||
`What are attention masks? <../glossary.html#attention-mask>`__
|
||||
token_type_ids (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`{0}`, `optional`, defaults to :obj:`None`):
|
||||
token_type_ids (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length)`, `optional`, defaults to :obj:`None`):
|
||||
Segment token indices to indicate first and second portions of the inputs.
|
||||
Indices are selected in ``[0, 1]``: ``0`` corresponds to a `sentence A` token, ``1``
|
||||
corresponds to a `sentence B` token
|
||||
|
||||
`What are token type IDs? <../glossary.html#token-type-ids>`_
|
||||
position_ids (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`{0}`, `optional`, defaults to :obj:`None`):
|
||||
position_ids (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length)`, `optional`, defaults to :obj:`None`):
|
||||
Indices of positions of each input sequence tokens in the position embeddings.
|
||||
Selected in the range ``[0, config.max_position_embeddings - 1]``.
|
||||
|
||||
@@ -681,15 +657,13 @@ ALBERT_INPUTS_DOCSTRING = r"""
|
||||
Mask to nullify selected heads of the self-attention modules.
|
||||
Mask values selected in ``[0, 1]``:
|
||||
``1`` indicates the head is **not masked**, ``0`` indicates the head is **masked**.
|
||||
inputs_embeds (:obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length, hidden_size)`, `optional`, defaults to :obj:`None`):
|
||||
input_embeds (:obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length, hidden_size)`, `optional`, defaults to :obj:`None`):
|
||||
Optionally, instead of passing :obj:`input_ids` you can choose to directly pass an embedded representation.
|
||||
This is useful if you want more control over how to convert `input_ids` indices into associated vectors
|
||||
than the model's internal embedding lookup matrix.
|
||||
training (:obj:`boolean`, `optional`, defaults to :obj:`False`):
|
||||
Whether to activate dropout modules (if set to :obj:`True`) during training or to de-activate them
|
||||
(if set to :obj:`False`) for evaluation.
|
||||
output_attentions (:obj:`bool`, `optional`, defaults to :obj:`None`):
|
||||
If set to ``True``, the attentions tensors of all attention layers are returned. See ``attentions`` under returned tensors for more detail.
|
||||
"""
|
||||
|
||||
|
||||
@@ -702,7 +676,7 @@ class TFAlbertModel(TFAlbertPreTrainedModel):
|
||||
super().__init__(config, *inputs, **kwargs)
|
||||
self.albert = TFAlbertMainLayer(config, name="albert")
|
||||
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING.format("(batch_size, sequence_length)"))
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING)
|
||||
def call(self, inputs, **kwargs):
|
||||
r"""
|
||||
Returns:
|
||||
@@ -721,7 +695,7 @@ class TFAlbertModel(TFAlbertPreTrainedModel):
|
||||
of shape :obj:`(batch_size, sequence_length, hidden_size)`.
|
||||
|
||||
Hidden-states of the model at the output of each layer plus the initial embedding outputs.
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``output_attentions=True`` is passed or ``config.output_attentions=True``):
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``config.output_attentions=True``):
|
||||
tuple of :obj:`tf.Tensor` (one for each layer) of shape
|
||||
:obj:`(batch_size, num_heads, sequence_length, sequence_length)`:
|
||||
|
||||
@@ -760,7 +734,7 @@ class TFAlbertForPreTraining(TFAlbertPreTrainedModel):
|
||||
def get_output_embeddings(self):
|
||||
return self.albert.embeddings
|
||||
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING.format("(batch_size, sequence_length)"))
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING)
|
||||
def call(self, inputs, **kwargs):
|
||||
r"""
|
||||
Return:
|
||||
@@ -773,7 +747,7 @@ class TFAlbertForPreTraining(TFAlbertPreTrainedModel):
|
||||
tuple of :obj:`tf.Tensor` (one for the output of the embeddings + one for the output of each layer)
|
||||
of shape :obj:`(batch_size, sequence_length, hidden_size)`.
|
||||
Hidden-states of the model at the output of each layer plus the initial embedding outputs.
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``output_attentions=True`` is passed or ``config.output_attentions=True``):
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``config.output_attentions=True``):
|
||||
tuple of :obj:`tf.Tensor` (one for each layer) of shape
|
||||
:obj:`(batch_size, num_heads, sequence_length, sequence_length)`:
|
||||
Attentions weights after the attention softmax, used to compute the weighted average in the self-attention heads.
|
||||
@@ -821,7 +795,7 @@ class TFAlbertForMaskedLM(TFAlbertPreTrainedModel):
|
||||
def get_output_embeddings(self):
|
||||
return self.albert.embeddings
|
||||
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING.format("(batch_size, sequence_length)"))
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING)
|
||||
def call(self, inputs, **kwargs):
|
||||
r"""
|
||||
Returns:
|
||||
@@ -833,7 +807,7 @@ class TFAlbertForMaskedLM(TFAlbertPreTrainedModel):
|
||||
of shape :obj:`(batch_size, sequence_length, hidden_size)`.
|
||||
|
||||
Hidden-states of the model at the output of each layer plus the initial embedding outputs.
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``output_attentions=True`` is passed or ``config.output_attentions=True``):
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``config.output_attentions=True``):
|
||||
tuple of :obj:`tf.Tensor` (one for each layer) of shape
|
||||
:obj:`(batch_size, num_heads, sequence_length, sequence_length)`:
|
||||
|
||||
@@ -867,7 +841,7 @@ class TFAlbertForMaskedLM(TFAlbertPreTrainedModel):
|
||||
the pooled output) e.g. for GLUE tasks. """,
|
||||
ALBERT_START_DOCSTRING,
|
||||
)
|
||||
class TFAlbertForSequenceClassification(TFAlbertPreTrainedModel, TFSequenceClassificationLoss):
|
||||
class TFAlbertForSequenceClassification(TFAlbertPreTrainedModel):
|
||||
def __init__(self, config, *inputs, **kwargs):
|
||||
super().__init__(config, *inputs, **kwargs)
|
||||
self.num_labels = config.num_labels
|
||||
@@ -879,25 +853,8 @@ class TFAlbertForSequenceClassification(TFAlbertPreTrainedModel, TFSequenceClass
|
||||
)
|
||||
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING)
|
||||
def call(
|
||||
self,
|
||||
input_ids=None,
|
||||
attention_mask=None,
|
||||
token_type_ids=None,
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
inputs_embeds=None,
|
||||
labels=None,
|
||||
output_attentions=None,
|
||||
training=False,
|
||||
):
|
||||
def call(self, inputs, **kwargs):
|
||||
r"""
|
||||
labels (:obj:`tf.Tensor` of shape :obj:`(batch_size,)`, `optional`, defaults to :obj:`None`):
|
||||
Labels for computing the sequence classification/regression loss.
|
||||
Indices should be in ``[0, ..., config.num_labels - 1]``.
|
||||
If ``config.num_labels == 1`` a regression loss is computed (Mean-Square loss),
|
||||
If ``config.num_labels > 1`` a classification loss is computed (Cross-Entropy).
|
||||
|
||||
Returns:
|
||||
:obj:`tuple(tf.Tensor)` comprising various elements depending on the configuration (:class:`~transformers.AlbertConfig`) and inputs:
|
||||
logits (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, config.num_labels)`)
|
||||
@@ -907,7 +864,7 @@ class TFAlbertForSequenceClassification(TFAlbertPreTrainedModel, TFSequenceClass
|
||||
of shape :obj:`(batch_size, sequence_length, hidden_size)`.
|
||||
|
||||
Hidden-states of the model at the output of each layer plus the initial embedding outputs.
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``output_attentions=True`` is passed or ``config.output_attentions=True``):
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``config.output_attentions=True``):
|
||||
tuple of :obj:`tf.Tensor` (one for each layer) of shape
|
||||
:obj:`(batch_size, num_heads, sequence_length, sequence_length)`:
|
||||
|
||||
@@ -921,129 +878,27 @@ class TFAlbertForSequenceClassification(TFAlbertPreTrainedModel, TFSequenceClass
|
||||
tokenizer = AlbertTokenizer.from_pretrained('albert-base-v2')
|
||||
model = TFAlbertForSequenceClassification.from_pretrained('albert-base-v2')
|
||||
input_ids = tf.constant(tokenizer.encode("Hello, my dog is cute"))[None, :] # Batch size 1
|
||||
labels = tf.reshape(tf.constant(1), (-1, 1)) # Batch size 1
|
||||
outputs = model(input_ids, labels=labels)
|
||||
loss, logits = outputs[:2]
|
||||
outputs = model(input_ids)
|
||||
logits = outputs[0]
|
||||
|
||||
"""
|
||||
|
||||
outputs = self.albert(
|
||||
input_ids,
|
||||
attention_mask=attention_mask,
|
||||
token_type_ids=token_type_ids,
|
||||
position_ids=position_ids,
|
||||
head_mask=head_mask,
|
||||
inputs_embeds=inputs_embeds,
|
||||
output_attentions=output_attentions,
|
||||
training=training,
|
||||
)
|
||||
outputs = self.albert(inputs, **kwargs)
|
||||
|
||||
pooled_output = outputs[1]
|
||||
|
||||
pooled_output = self.dropout(pooled_output, training=training)
|
||||
pooled_output = self.dropout(pooled_output, training=kwargs.get("training", False))
|
||||
logits = self.classifier(pooled_output)
|
||||
|
||||
outputs = (logits,) + outputs[2:] # add hidden states and attention if they are here
|
||||
|
||||
if labels is not None:
|
||||
loss = self.compute_loss(labels, logits)
|
||||
outputs = (loss,) + outputs
|
||||
|
||||
return outputs # (loss), logits, (hidden_states), (attentions)
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
"""Albert Model with a token classification head on top (a linear layer on top of
|
||||
the hidden-states output) e.g. for Named-Entity-Recognition (NER) tasks. """,
|
||||
ALBERT_START_DOCSTRING,
|
||||
)
|
||||
class TFAlbertForTokenClassification(TFAlbertPreTrainedModel, TFTokenClassificationLoss):
|
||||
def __init__(self, config, *inputs, **kwargs):
|
||||
super().__init__(config, *inputs, **kwargs)
|
||||
self.num_labels = config.num_labels
|
||||
|
||||
self.albert = TFAlbertMainLayer(config, name="albert")
|
||||
self.dropout = tf.keras.layers.Dropout(config.hidden_dropout_prob)
|
||||
self.classifier = tf.keras.layers.Dense(
|
||||
config.num_labels, kernel_initializer=get_initializer(config.initializer_range), name="classifier"
|
||||
)
|
||||
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING)
|
||||
def call(
|
||||
self,
|
||||
input_ids=None,
|
||||
attention_mask=None,
|
||||
token_type_ids=None,
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
inputs_embeds=None,
|
||||
labels=None,
|
||||
output_attentions=None,
|
||||
training=False,
|
||||
):
|
||||
r"""
|
||||
labels (:obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length)`, `optional`, defaults to :obj:`None`):
|
||||
Labels for computing the token classification loss.
|
||||
Indices should be in ``[0, ..., config.num_labels - 1]``.
|
||||
|
||||
Return:
|
||||
:obj:`tuple(tf.Tensor)` comprising various elements depending on the configuration (:class:`~transformers.BertConfig`) and inputs:
|
||||
scores (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length, config.num_labels)`):
|
||||
Classification scores (before SoftMax).
|
||||
hidden_states (:obj:`tuple(tf.Tensor)`, `optional`, returned when :obj:`config.output_hidden_states=True`):
|
||||
tuple of :obj:`tf.Tensor` (one for the output of the embeddings + one for the output of each layer)
|
||||
of shape :obj:`(batch_size, sequence_length, hidden_size)`.
|
||||
|
||||
Hidden-states of the model at the output of each layer plus the initial embedding outputs.
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``output_attentions=True`` is passed or ``config.output_attentions=True``):
|
||||
tuple of :obj:`tf.Tensor` (one for each layer) of shape
|
||||
:obj:`(batch_size, num_heads, sequence_length, sequence_length)`:
|
||||
|
||||
Attentions weights after the attention softmax, used to compute the weighted average in the self-attention heads.
|
||||
|
||||
Examples::
|
||||
|
||||
import tensorflow as tf
|
||||
from transformers import AlbertTokenizer, TFAlbertForTokenClassification
|
||||
|
||||
tokenizer = AlbertTokenizer.from_pretrained('albert-base-v2')
|
||||
model = TFAlbertForTokenClassification.from_pretrained('albert-base-v2')
|
||||
input_ids = tf.constant(tokenizer.encode("Hello, my dog is cute", add_special_tokens=True))[None, :] # Batch size 1
|
||||
labels = tf.reshape(tf.constant([1] * tf.size(input_ids).numpy()), (-1, tf.size(input_ids))) # Batch size 1
|
||||
outputs = model(input_ids, labels=labels)
|
||||
loss, scores = outputs[:2]
|
||||
|
||||
"""
|
||||
outputs = self.albert(
|
||||
input_ids,
|
||||
attention_mask=attention_mask,
|
||||
token_type_ids=token_type_ids,
|
||||
position_ids=position_ids,
|
||||
head_mask=head_mask,
|
||||
inputs_embeds=inputs_embeds,
|
||||
output_attentions=output_attentions,
|
||||
training=training,
|
||||
)
|
||||
|
||||
sequence_output = outputs[0]
|
||||
|
||||
sequence_output = self.dropout(sequence_output, training=training)
|
||||
logits = self.classifier(sequence_output)
|
||||
|
||||
outputs = (logits,) + outputs[2:] # add hidden states and attention if they are here
|
||||
|
||||
if labels is not None:
|
||||
loss = self.compute_loss(labels, logits)
|
||||
outputs = (loss,) + outputs
|
||||
|
||||
return outputs # (loss), logits, (hidden_states), (attentions)
|
||||
return outputs # logits, (hidden_states), (attentions)
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
"""Albert Model with a span classification head on top for extractive question-answering tasks like SQuAD (a linear layers on top of the hidden-states output to compute `span start logits` and `span end logits`). """,
|
||||
ALBERT_START_DOCSTRING,
|
||||
)
|
||||
class TFAlbertForQuestionAnswering(TFAlbertPreTrainedModel, TFQuestionAnsweringLoss):
|
||||
class TFAlbertForQuestionAnswering(TFAlbertPreTrainedModel):
|
||||
def __init__(self, config, *inputs, **kwargs):
|
||||
super().__init__(config, *inputs, **kwargs)
|
||||
self.num_labels = config.num_labels
|
||||
@@ -1054,32 +909,8 @@ class TFAlbertForQuestionAnswering(TFAlbertPreTrainedModel, TFQuestionAnsweringL
|
||||
)
|
||||
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING)
|
||||
def call(
|
||||
self,
|
||||
input_ids=None,
|
||||
attention_mask=None,
|
||||
token_type_ids=None,
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
inputs_embeds=None,
|
||||
start_positions=None,
|
||||
end_positions=None,
|
||||
cls_index=None,
|
||||
p_mask=None,
|
||||
is_impossible=None,
|
||||
output_attentions=None,
|
||||
training=False,
|
||||
):
|
||||
def call(self, inputs, **kwargs):
|
||||
r"""
|
||||
start_positions (:obj:`tf.Tensor` of shape :obj:`(batch_size,)`, `optional`, defaults to :obj:`None`):
|
||||
Labels for position (index) of the start of the labelled span for computing the token classification loss.
|
||||
Positions are clamped to the length of the sequence (`sequence_length`).
|
||||
Position outside of the sequence are not taken into account for computing the loss.
|
||||
end_positions (:obj:`tf.Tensor` of shape :obj:`(batch_size,)`, `optional`, defaults to :obj:`None`):
|
||||
Labels for position (index) of the end of the labelled span for computing the token classification loss.
|
||||
Positions are clamped to the length of the sequence (`sequence_length`).
|
||||
Position outside of the sequence are not taken into account for computing the loss.
|
||||
|
||||
Return:
|
||||
:obj:`tuple(tf.Tensor)` comprising various elements depending on the configuration (:class:`~transformers.AlbertConfig`) and inputs:
|
||||
start_scores (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, sequence_length,)`):
|
||||
@@ -1091,7 +922,7 @@ class TFAlbertForQuestionAnswering(TFAlbertPreTrainedModel, TFQuestionAnsweringL
|
||||
of shape :obj:`(batch_size, sequence_length, hidden_size)`.
|
||||
|
||||
Hidden-states of the model at the output of each layer plus the initial embedding outputs.
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``output_attentions=True`` is passed or ``config.output_attentions=True``):
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``config.output_attentions=True``):
|
||||
tuple of :obj:`tf.Tensor` (one for each layer) of shape
|
||||
:obj:`(batch_size, num_heads, sequence_length, sequence_length)`:
|
||||
|
||||
@@ -1107,24 +938,14 @@ class TFAlbertForQuestionAnswering(TFAlbertPreTrainedModel, TFQuestionAnsweringL
|
||||
|
||||
tokenizer = AlbertTokenizer.from_pretrained('albert-base-v2')
|
||||
model = TFAlbertForQuestionAnswering.from_pretrained('albert-base-v2')
|
||||
question, text = "Who was Jim Henson?", "Jim Henson was a nice puppet"
|
||||
input_dict = tokenizer.encode_plus(question, text, return_tensors='tf')
|
||||
start_scores, end_scores = model(input_dict)
|
||||
input_ids = tokenizer.encode("Who was Jim Henson?", "Jim Henson was a nice puppet")
|
||||
start_scores, end_scores = model(tf.constant(input_ids)[None, :]) # Batch size 1
|
||||
|
||||
all_tokens = tokenizer.convert_ids_to_tokens(input_dict["input_ids"].numpy()[0])
|
||||
all_tokens = tokenizer.convert_ids_to_tokens(input_ids)
|
||||
answer = ' '.join(all_tokens[tf.math.argmax(start_scores, 1)[0] : tf.math.argmax(end_scores, 1)[0]+1])
|
||||
|
||||
"""
|
||||
outputs = self.albert(
|
||||
input_ids,
|
||||
attention_mask=attention_mask,
|
||||
token_type_ids=token_type_ids,
|
||||
position_ids=position_ids,
|
||||
head_mask=head_mask,
|
||||
inputs_embeds=inputs_embeds,
|
||||
output_attentions=output_attentions,
|
||||
training=training,
|
||||
)
|
||||
outputs = self.albert(inputs, **kwargs)
|
||||
|
||||
sequence_output = outputs[0]
|
||||
|
||||
@@ -1135,13 +956,7 @@ class TFAlbertForQuestionAnswering(TFAlbertPreTrainedModel, TFQuestionAnsweringL
|
||||
|
||||
outputs = (start_logits, end_logits,) + outputs[2:]
|
||||
|
||||
if start_positions is not None and end_positions is not None:
|
||||
labels = {"start_position": start_positions}
|
||||
labels["end_position"] = end_positions
|
||||
loss = self.compute_loss(labels, outputs[:2])
|
||||
outputs = (loss,) + outputs
|
||||
|
||||
return outputs # (loss), start_logits, end_logits, (hidden_states), (attentions)
|
||||
return outputs # start_logits, end_logits, (hidden_states), (attentions)
|
||||
|
||||
|
||||
@add_start_docstrings(
|
||||
@@ -1149,7 +964,7 @@ class TFAlbertForQuestionAnswering(TFAlbertPreTrainedModel, TFQuestionAnsweringL
|
||||
the pooled output and a softmax) e.g. for RocStories/SWAG tasks. """,
|
||||
ALBERT_START_DOCSTRING,
|
||||
)
|
||||
class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel):
|
||||
def __init__(self, config, *inputs, **kwargs):
|
||||
super().__init__(config, *inputs, **kwargs)
|
||||
|
||||
@@ -1168,7 +983,7 @@ class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
"""
|
||||
return {"input_ids": tf.constant(MULTIPLE_CHOICE_DUMMY_INPUTS)}
|
||||
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING.format("(batch_size, num_choices, sequence_length)"))
|
||||
@add_start_docstrings_to_callable(ALBERT_INPUTS_DOCSTRING)
|
||||
def call(
|
||||
self,
|
||||
inputs,
|
||||
@@ -1177,16 +992,9 @@ class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
inputs_embeds=None,
|
||||
labels=None,
|
||||
output_attentions=None,
|
||||
training=False,
|
||||
):
|
||||
r"""
|
||||
labels (:obj:`tf.Tensor` of shape :obj:`(batch_size,)`, `optional`, defaults to :obj:`None`):
|
||||
Labels for computing the multiple choice classification loss.
|
||||
Indices should be in ``[0, ..., num_choices]`` where `num_choices` is the size of the second dimension
|
||||
of the input tensors. (see `input_ids` above)
|
||||
|
||||
Return:
|
||||
:obj:`tuple(tf.Tensor)` comprising various elements depending on the configuration (:class:`~transformers.BertConfig`) and inputs:
|
||||
classification_scores (:obj:`Numpy array` or :obj:`tf.Tensor` of shape :obj:`(batch_size, num_choices)`:
|
||||
@@ -1198,7 +1006,7 @@ class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
of shape :obj:`(batch_size, sequence_length, hidden_size)`.
|
||||
|
||||
Hidden-states of the model at the output of each layer plus the initial embedding outputs.
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``output_attentions=True`` is passed or ``config.output_attentions=True``):
|
||||
attentions (:obj:`tuple(tf.Tensor)`, `optional`, returned when ``config.output_attentions=True``):
|
||||
tuple of :obj:`tf.Tensor` (one for each layer) of shape
|
||||
:obj:`(batch_size, num_heads, sequence_length, sequence_length)`:
|
||||
|
||||
@@ -1211,13 +1019,12 @@ class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
|
||||
tokenizer = AlbertTokenizer.from_pretrained('albert-base-v2')
|
||||
model = TFAlbertForMultipleChoice.from_pretrained('albert-base-v2')
|
||||
choices = ["Hello, my dog is cute", "Hello, my cat is amazing"]
|
||||
|
||||
input_ids = tokenizer(choices, add_special_tokens=True, return_tensors='tf', truncation=True, padding=True)[None, :] # Batch size 1, 2 choices
|
||||
labels = tf.reshape(tf.constant(1), (-1, 1))
|
||||
outputs = model(input_ids, labels=labels)
|
||||
|
||||
loss, classification_scores = outputs[:2]
|
||||
example1 = ["This is a context", "Is it a context? Yes"]
|
||||
example2 = ["This is a context", "Is it a context? No"]
|
||||
encoding = tokenizer.batch_encode_plus([example1, example2], return_tensors='tf', truncation=True, padding=True)
|
||||
outputs = model(encoding["input_ids"][None, :])
|
||||
logits = outputs[0]
|
||||
|
||||
"""
|
||||
if isinstance(inputs, (tuple, list)):
|
||||
@@ -1227,17 +1034,18 @@ class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
position_ids = inputs[3] if len(inputs) > 3 else position_ids
|
||||
head_mask = inputs[4] if len(inputs) > 4 else head_mask
|
||||
inputs_embeds = inputs[5] if len(inputs) > 5 else inputs_embeds
|
||||
output_attentions = inputs[6] if len(inputs) > 6 else output_attentions
|
||||
assert len(inputs) <= 7, "Too many inputs."
|
||||
assert len(inputs) <= 6, "Too many inputs."
|
||||
elif isinstance(inputs, dict):
|
||||
print("isdict(1)")
|
||||
input_ids = inputs.get("input_ids")
|
||||
print(input_ids)
|
||||
|
||||
attention_mask = inputs.get("attention_mask", attention_mask)
|
||||
token_type_ids = inputs.get("token_type_ids", token_type_ids)
|
||||
position_ids = inputs.get("position_ids", position_ids)
|
||||
head_mask = inputs.get("head_mask", head_mask)
|
||||
inputs_embeds = inputs.get("inputs_embeds", inputs_embeds)
|
||||
output_attentions = inputs.get("output_attentions", output_attentions)
|
||||
assert len(inputs) <= 7, "Too many inputs."
|
||||
assert len(inputs) <= 6, "Too many inputs."
|
||||
else:
|
||||
input_ids = inputs
|
||||
|
||||
@@ -1260,7 +1068,6 @@ class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
flat_position_ids,
|
||||
head_mask,
|
||||
inputs_embeds,
|
||||
output_attentions,
|
||||
]
|
||||
|
||||
outputs = self.albert(flat_inputs, training=training)
|
||||
@@ -1273,8 +1080,4 @@ class TFAlbertForMultipleChoice(TFAlbertPreTrainedModel, TFMultipleChoiceLoss):
|
||||
|
||||
outputs = (reshaped_logits,) + outputs[2:] # add hidden states and attention if they are here
|
||||
|
||||
if labels is not None:
|
||||
loss = self.compute_loss(labels, reshaped_logits)
|
||||
outputs = (loss,) + outputs
|
||||
|
||||
return outputs # (loss), reshaped_logits, (hidden_states), (attentions)
|
||||
return outputs # reshaped_logits, (hidden_states), (attentions)
|
||||
|
||||
@@ -21,7 +21,7 @@ import logging
|
||||
import re
|
||||
from typing import List, Optional, Tuple, Union
|
||||
|
||||
from .file_utils import add_end_docstrings
|
||||
from .file_utils import add_end_docstrings, is_tf_available, is_torch_available
|
||||
from .tokenization_utils_base import (
|
||||
ENCODE_KWARGS_DOCSTRING,
|
||||
ENCODE_PLUS_ADDITIONAL_KWARGS_DOCSTRING,
|
||||
@@ -32,13 +32,17 @@ from .tokenization_utils_base import (
|
||||
PreTokenizedInput,
|
||||
PreTokenizedInputPair,
|
||||
PreTrainedTokenizerBase,
|
||||
TensorType,
|
||||
TextInput,
|
||||
TextInputPair,
|
||||
TruncationStrategy,
|
||||
)
|
||||
|
||||
|
||||
if is_tf_available():
|
||||
import tensorflow as tf
|
||||
if is_torch_available():
|
||||
import torch
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -54,19 +58,19 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
|
||||
Class attributes (overridden by derived classes):
|
||||
|
||||
- ``vocab_files_names``: a python ``dict`` with, as keys, the ``__init__`` keyword name of each vocabulary file
|
||||
required by the model, and as associated values, the filename for saving the associated file (string).
|
||||
- ``pretrained_vocab_files_map``: a python ``dict of dict`` the high-level keys
|
||||
being the ``__init__`` keyword name of each vocabulary file required by the model, the low-level being the
|
||||
`short-cut-names` (string) of the pretrained models with, as associated values, the `url` (string) to the
|
||||
associated pretrained vocabulary file.
|
||||
- ``max_model_input_sizes``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the pretrained
|
||||
models, and as associated values, the maximum length of the sequence inputs of this model, or None if the
|
||||
model has no maximum input size.
|
||||
- ``pretrained_init_configuration``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the
|
||||
pretrained models, and as associated values, a dictionnary of specific arguments to pass to the
|
||||
``__init__``method of the tokenizer class for this pretrained model when loading the tokenizer with the
|
||||
``from_pretrained()`` method.
|
||||
- ``vocab_files_names``: a python ``dict`` with, as keys, the ``__init__`` keyword name of each vocabulary file
|
||||
required by the model, and as associated values, the filename for saving the associated file (string).
|
||||
- ``pretrained_vocab_files_map``: a python ``dict of dict`` the high-level keys
|
||||
being the ``__init__`` keyword name of each vocabulary file required by the model, the low-level being the
|
||||
`short-cut-names` (string) of the pretrained models with, as associated values, the `url` (string) to the
|
||||
associated pretrained vocabulary file.
|
||||
- ``max_model_input_sizes``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the pretrained
|
||||
models, and as associated values, the maximum length of the sequence inputs of this model, or None if the
|
||||
model has no maximum input size.
|
||||
- ``pretrained_init_configuration``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the
|
||||
pretrained models, and as associated values, a dictionnary of specific arguments to pass to the
|
||||
``__init__``method of the tokenizer class for this pretrained model when loading the tokenizer with the
|
||||
``from_pretrained()`` method.
|
||||
|
||||
Args:
|
||||
- ``model_max_length``: (`Optional`) int: the maximum length in number of tokens for the inputs to the transformer model.
|
||||
@@ -95,16 +99,13 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
- ``additional_special_tokens``: (`Optional`) list: a list of additional special tokens.
|
||||
Adding all special tokens here ensure they won't be split by the tokenization process.
|
||||
Will be associated to ``self.additional_special_tokens`` and ``self.additional_special_tokens_ids``
|
||||
|
||||
|
||||
.. automethod:: __call__
|
||||
"""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
# Added tokens
|
||||
self.added_tokens_encoder = {}
|
||||
self.unique_added_tokens_encoder = set()
|
||||
self.unique_added_tokens_encoder = []
|
||||
self.added_tokens_decoder = {}
|
||||
|
||||
@property
|
||||
@@ -125,27 +126,6 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
return self.vocab_size + len(self.added_tokens_encoder)
|
||||
|
||||
def add_tokens(self, new_tokens: Union[str, List[str]]) -> int:
|
||||
"""
|
||||
Add a list of new tokens to the tokenizer class. If the new tokens are not in the
|
||||
vocabulary, they are added to it with indices starting from length of the current vocabulary.
|
||||
|
||||
Args:
|
||||
new_tokens: string or list of string. Each string is a token to add. Tokens are only added if they are not
|
||||
already in the vocabulary (tested by checking if the tokenizer assign the index of the ``unk_token`` to them).
|
||||
|
||||
Returns:
|
||||
Number of tokens added to the vocabulary.
|
||||
|
||||
Examples::
|
||||
|
||||
# Let's see how to increase the vocabulary of Bert model and tokenizer
|
||||
tokenizer = BertTokenizer.from_pretrained('bert-base-uncased')
|
||||
model = BertModel.from_pretrained('bert-base-uncased')
|
||||
|
||||
num_added_toks = tokenizer.add_tokens(['new_tok1', 'my_new-tok2'])
|
||||
print('We have added', num_added_toks, 'tokens')
|
||||
model.resize_token_embeddings(len(tokenizer)) # Notice: resize_token_embeddings expect to receive the full size of the new vocabulary, i.e. the length of the tokenizer.
|
||||
"""
|
||||
if not new_tokens:
|
||||
return 0
|
||||
|
||||
@@ -169,7 +149,8 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
added_tok_encoder = dict((tok, len(self) + i) for i, tok in enumerate(tokens_to_add))
|
||||
added_tok_decoder = {v: k for k, v in added_tok_encoder.items()}
|
||||
self.added_tokens_encoder.update(added_tok_encoder)
|
||||
self.unique_added_tokens_encoder = set(self.added_tokens_encoder.keys()).union(set(self.all_special_tokens))
|
||||
# we don't store a set because they are not deterministic with pickle/dill and it messes up with HuggingFace nlp library caching
|
||||
self.unique_added_tokens_encoder = sorted(list(set(list(self.added_tokens_encoder.keys()) + self.all_special_tokens)))
|
||||
self.added_tokens_decoder.update(added_tok_decoder)
|
||||
|
||||
return len(tokens_to_add)
|
||||
@@ -310,7 +291,7 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
is_pretokenized: bool = False,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_mask: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -364,7 +345,6 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
max_length=max_length,
|
||||
stride=stride,
|
||||
return_tensors=return_tensors,
|
||||
prepend_batch_axis=True,
|
||||
return_attention_mask=return_attention_mask,
|
||||
return_token_type_ids=return_token_type_ids,
|
||||
return_overflowing_tokens=return_overflowing_tokens,
|
||||
@@ -388,7 +368,7 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
is_pretokenized: bool = False,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_masks: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -454,12 +434,44 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
return_overflowing_tokens=return_overflowing_tokens,
|
||||
return_special_tokens_masks=return_special_tokens_masks,
|
||||
return_lengths=return_lengths,
|
||||
return_tensors=return_tensors,
|
||||
return_tensors=None, # We will convert the whole batch to tensors at the end
|
||||
verbose=verbose,
|
||||
)
|
||||
|
||||
if return_tensors is not None:
|
||||
self.convert_to_tensors_(batch_outputs, return_tensors, verbose=verbose)
|
||||
|
||||
return BatchEncoding(batch_outputs)
|
||||
|
||||
def convert_to_tensors_(self, batch_outputs: dict, return_tensors: str, verbose: bool = True) -> None:
|
||||
# Do the tensor conversion in batch
|
||||
for key, value in batch_outputs.items():
|
||||
if return_tensors == "tf" and is_tf_available():
|
||||
try:
|
||||
batch_outputs[key] = tf.constant(value)
|
||||
except ValueError:
|
||||
if None in [item for sequence in value for item in sequence]:
|
||||
raise ValueError(self.NO_PAD_TOKEN_FOR_BATCH_MSG)
|
||||
else:
|
||||
raise ValueError(self.UNEVEN_SEQUENCES_FOR_BATCH_MSG)
|
||||
elif return_tensors == "pt" and is_torch_available():
|
||||
try:
|
||||
batch_outputs[key] = torch.tensor(value)
|
||||
except ValueError:
|
||||
raise ValueError(self.UNEVEN_SEQUENCES_FOR_BATCH_MSG)
|
||||
except RuntimeError:
|
||||
if None in [item for sequence in value for item in sequence]:
|
||||
raise ValueError(self.NO_PAD_TOKEN_FOR_BATCH_MSG)
|
||||
else:
|
||||
raise
|
||||
|
||||
elif return_tensors is not None and verbose:
|
||||
logger.warning(
|
||||
"Unable to convert output to tensors format {}, PyTorch or TensorFlow is not available.".format(
|
||||
return_tensors
|
||||
)
|
||||
)
|
||||
|
||||
@add_end_docstrings(ENCODE_KWARGS_DOCSTRING, ENCODE_PLUS_ADDITIONAL_KWARGS_DOCSTRING)
|
||||
def _batch_prepare_for_model(
|
||||
self,
|
||||
@@ -513,7 +525,6 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
return_special_tokens_mask=return_special_tokens_masks,
|
||||
return_lengths=return_lengths,
|
||||
return_tensors=None, # We will convert the whole batch to tensors at the end
|
||||
prepend_batch_axis=False,
|
||||
verbose=verbose,
|
||||
)
|
||||
|
||||
@@ -522,8 +533,6 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
batch_outputs[key] = []
|
||||
batch_outputs[key].append(value)
|
||||
|
||||
batch_outputs = BatchEncoding(batch_outputs, tensor_type=return_tensors)
|
||||
|
||||
return batch_outputs
|
||||
|
||||
@add_end_docstrings(ENCODE_KWARGS_DOCSTRING, ENCODE_PLUS_ADDITIONAL_KWARGS_DOCSTRING)
|
||||
@@ -537,7 +546,6 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
return_tensors: Optional[str] = None,
|
||||
prepend_batch_axis: bool = False,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_mask: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -618,11 +626,32 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
if return_lengths:
|
||||
encoded_inputs["length"] = len(encoded_inputs["input_ids"])
|
||||
|
||||
batch_outputs = BatchEncoding(
|
||||
encoded_inputs, tensor_type=return_tensors, prepend_batch_axis=prepend_batch_axis
|
||||
)
|
||||
# Prepare model inputs as tensors if asked
|
||||
if return_tensors == "tf" and is_tf_available():
|
||||
encoded_inputs["input_ids"] = tf.constant([encoded_inputs["input_ids"]])
|
||||
|
||||
return batch_outputs
|
||||
if "token_type_ids" in encoded_inputs:
|
||||
encoded_inputs["token_type_ids"] = tf.constant([encoded_inputs["token_type_ids"]])
|
||||
|
||||
if "attention_mask" in encoded_inputs:
|
||||
encoded_inputs["attention_mask"] = tf.constant([encoded_inputs["attention_mask"]])
|
||||
|
||||
elif return_tensors == "pt" and is_torch_available():
|
||||
encoded_inputs["input_ids"] = torch.tensor([encoded_inputs["input_ids"]])
|
||||
|
||||
if "token_type_ids" in encoded_inputs:
|
||||
encoded_inputs["token_type_ids"] = torch.tensor([encoded_inputs["token_type_ids"]])
|
||||
|
||||
if "attention_mask" in encoded_inputs:
|
||||
encoded_inputs["attention_mask"] = torch.tensor([encoded_inputs["attention_mask"]])
|
||||
elif return_tensors is not None and verbose:
|
||||
logger.warning(
|
||||
"Unable to convert output to tensors format {}, PyTorch or TensorFlow is not available.".format(
|
||||
return_tensors
|
||||
)
|
||||
)
|
||||
|
||||
return BatchEncoding(encoded_inputs)
|
||||
|
||||
def prepare_for_tokenization(self, text: str, **kwargs) -> str:
|
||||
""" Performs any necessary transformations before tokenization """
|
||||
@@ -645,14 +674,12 @@ class PreTrainedTokenizer(PreTrainedTokenizerBase):
|
||||
`tokenize` and `convert_tokens_to_ids` methods.
|
||||
num_tokens_to_remove (:obj:`int`, `optional`, defaults to ``0``):
|
||||
number of tokens to remove using the truncation strategy
|
||||
truncation_strategy (:obj:`string`, `optional`, defaults to "only_first"):
|
||||
String selected in the following options:
|
||||
|
||||
truncation_strategy: string selected in the following options:
|
||||
- 'only_first' (default): Only truncate the first sequence. raise an error if the first sequence is shorter or equal to than num_tokens_to_remove.
|
||||
- 'only_second': Only truncate the second sequence
|
||||
- 'longest_first': Iteratively reduce the inputs sequence until the input is under max_length
|
||||
starting from the longest one at each token (when there is a pair of input sequences).
|
||||
Overflowing tokens only contains overflow from the first sequence.
|
||||
- 'longest_first' Iteratively reduce the inputs sequence until the input is under max_length
|
||||
starting from the longest one at each token (when there is a pair of input sequences).
|
||||
Overflowing tokens only contains overflow from the first sequence.
|
||||
- 'do_not_truncate'
|
||||
stride (:obj:`int`, `optional`, defaults to ``0``):
|
||||
If set to a number along with max_length, the overflowing tokens returned will contain some tokens
|
||||
|
||||
@@ -27,25 +27,10 @@ from collections import UserDict
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, List, NamedTuple, Optional, Sequence, Tuple, Union
|
||||
|
||||
import numpy as np
|
||||
from tokenizers import AddedToken as AddedTokenFast
|
||||
from tokenizers import Encoding as EncodingFast
|
||||
|
||||
from .file_utils import (
|
||||
add_end_docstrings,
|
||||
cached_path,
|
||||
hf_bucket_url,
|
||||
is_remote_url,
|
||||
is_tf_available,
|
||||
is_torch_available,
|
||||
torch_required,
|
||||
)
|
||||
|
||||
|
||||
if is_tf_available():
|
||||
import tensorflow as tf
|
||||
if is_torch_available():
|
||||
import torch
|
||||
from .file_utils import add_end_docstrings, cached_path, hf_bucket_url, is_remote_url, torch_required
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -62,14 +47,17 @@ PreTokenizedInputPair = Tuple[List[str], List[str]]
|
||||
EncodedInputPair = Tuple[List[int], List[int]]
|
||||
|
||||
|
||||
# Slow tokenizers used to be saved in three separated files
|
||||
SPECIAL_TOKENS_MAP_FILE = "special_tokens_map.json"
|
||||
ADDED_TOKENS_FILE = "added_tokens.json"
|
||||
TOKENIZER_CONFIG_FILE = "tokenizer_config.json"
|
||||
|
||||
# Fast tokenizers (provided by HuggingFace tokenizer's library) can be saved in a single file
|
||||
FULL_TOKENIZER_FILE = "tokenizer.json"
|
||||
|
||||
|
||||
class ExplicitEnum(Enum):
|
||||
""" Enum with more explicit error message for missing values.
|
||||
""" With more explicit missing values error message.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
@@ -93,12 +81,6 @@ class PaddingStrategy(ExplicitEnum):
|
||||
DO_NOT_PAD = "do_not_pad"
|
||||
|
||||
|
||||
class TensorType(ExplicitEnum):
|
||||
PYTORCH = "pt"
|
||||
TENSORFLOW = "tf"
|
||||
NUMPY = "np"
|
||||
|
||||
|
||||
class CharSpan(NamedTuple):
|
||||
""" Character span in the original string
|
||||
|
||||
@@ -133,18 +115,13 @@ class BatchEncoding(UserDict):
|
||||
encoding (:obj:`EncodingFast`, :obj:`list(EncodingFast)`, `optional`, defaults to :obj:`None`):
|
||||
If the tokenizer is a fast tokenizer which outputs additional informations like mapping from word/char space to token space
|
||||
the `EncodingFast` instance or list of instance (for batches) hold these informations.
|
||||
tensor_type (:obj:`Union[None, str, TensorType]`, `optional`, defaults to :obj:`None`):
|
||||
You can give a tensor_type here to convert the lists of integers in PyTorch/TF/Numpy Tensors at initialization
|
||||
prepend_batch_axis (:obj:`bool`, `optional`, defaults to :obj:`False`):
|
||||
Set to True to add a batch axis when converting in Tensors (see :obj:`tensor_type` above)
|
||||
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
data: Optional[Dict[str, Any]] = None,
|
||||
encoding: Optional[Union[EncodingFast, Sequence[EncodingFast]]] = None,
|
||||
tensor_type: Union[None, str, TensorType] = None,
|
||||
prepend_batch_axis: bool = False,
|
||||
):
|
||||
super().__init__(data)
|
||||
|
||||
@@ -153,16 +130,6 @@ class BatchEncoding(UserDict):
|
||||
|
||||
self._encodings = encoding
|
||||
|
||||
self.convert_to_tensors(tensor_type=tensor_type, prepend_batch_axis=prepend_batch_axis)
|
||||
|
||||
@property
|
||||
def is_fast(self):
|
||||
"""
|
||||
Indicate if this BatchEncoding was generated from the result of a PreTrainedTokenizerFast
|
||||
Returns: True if generated from subclasses of PreTrainedTokenizerFast, else otherwise
|
||||
"""
|
||||
return self._encodings is not None
|
||||
|
||||
def __getitem__(self, item: Union[int, str]) -> EncodingFast:
|
||||
""" If the key is a string, get the value of the dict associated to `key` ('input_ids', 'attention_mask'...)
|
||||
If the key is an integer, get the EncodingFast for batch item with index `key`
|
||||
@@ -178,20 +145,7 @@ class BatchEncoding(UserDict):
|
||||
)
|
||||
|
||||
def __getattr__(self, item: str):
|
||||
try:
|
||||
return self.data[item]
|
||||
except KeyError:
|
||||
raise AttributeError
|
||||
|
||||
def __getstate__(self):
|
||||
return {"data": self.data, "encodings": self._encodings}
|
||||
|
||||
def __setstate__(self, state):
|
||||
if "data" in state:
|
||||
self.data = state["data"]
|
||||
|
||||
if "encodings" in state:
|
||||
self._encodings = state["encodings"]
|
||||
return self.data[item]
|
||||
|
||||
def keys(self):
|
||||
return self.data.keys()
|
||||
@@ -215,7 +169,7 @@ class BatchEncoding(UserDict):
|
||||
"""
|
||||
return self._encodings
|
||||
|
||||
def tokens(self, batch_index: int = 0) -> List[str]:
|
||||
def tokens(self, batch_index: int = 0) -> List[int]:
|
||||
if not self._encodings:
|
||||
raise ValueError("tokens() is not available when using Python based tokenizers")
|
||||
return self._encodings[batch_index].tokens
|
||||
@@ -226,18 +180,16 @@ class BatchEncoding(UserDict):
|
||||
return self._encodings[batch_index].words
|
||||
|
||||
def token_to_word(self, batch_or_token_index: int, token_index: Optional[int] = None) -> int:
|
||||
"""
|
||||
Get the index of the word corresponding (i.e. comprising) to an encoded token
|
||||
in a sequence of the batch.
|
||||
""" Get the index of the word corresponding (i.e. comprising) to an encoded token
|
||||
in a sequence of the batch.
|
||||
|
||||
Can be called as:
|
||||
Can be called as:
|
||||
- self.token_to_word(token_index) if batch size is 1
|
||||
- self.token_to_word(batch_index, token_index) if batch size is greater than 1
|
||||
|
||||
- ``self.token_to_word(token_index)`` if batch size is 1
|
||||
- ``self.token_to_word(batch_index, token_index)`` if batch size is greater than 1
|
||||
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
|
||||
Args:
|
||||
batch_or_token_index (:obj:`int`):
|
||||
@@ -248,7 +200,7 @@ class BatchEncoding(UserDict):
|
||||
of the token in the sequence.
|
||||
|
||||
Returns:
|
||||
:obj:`int`:
|
||||
word_index (:obj:`int`):
|
||||
index of the word in the input sequence.
|
||||
|
||||
"""
|
||||
@@ -267,22 +219,19 @@ class BatchEncoding(UserDict):
|
||||
return self._encodings[batch_index].token_to_word(token_index)
|
||||
|
||||
def word_to_tokens(self, batch_or_word_index: int, word_index: Optional[int] = None) -> TokenSpan:
|
||||
"""
|
||||
Get the encoded token span corresponding to a word in the sequence of the batch.
|
||||
""" Get the encoded token span corresponding to a word in the sequence of the batch.
|
||||
|
||||
Token spans are returned as a TokenSpan NamedTuple with:
|
||||
Token spans are returned as a TokenSpan NamedTuple with:
|
||||
start: index of the first token
|
||||
end: index of the token following the last token
|
||||
|
||||
- start: index of the first token
|
||||
- end: index of the token following the last token
|
||||
Can be called as:
|
||||
- self.word_to_tokens(word_index) if batch size is 1
|
||||
- self.word_to_tokens(batch_index, word_index) if batch size is greater or equal to 1
|
||||
|
||||
Can be called as:
|
||||
|
||||
- ``self.word_to_tokens(word_index)`` if batch size is 1
|
||||
- ``self.word_to_tokens(batch_index, word_index)`` if batch size is greater or equal to 1
|
||||
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
|
||||
Args:
|
||||
batch_or_word_index (:obj:`int`):
|
||||
@@ -293,13 +242,12 @@ class BatchEncoding(UserDict):
|
||||
of the word in the sequence.
|
||||
|
||||
Returns:
|
||||
:obj:`TokenSpan`:
|
||||
token_span (:obj:`TokenSpan`):
|
||||
Span of tokens in the encoded sequence.
|
||||
|
||||
:obj:`TokenSpan` are NamedTuple with:
|
||||
|
||||
- start: index of the first token
|
||||
- end: index of the token following the last token
|
||||
TokenSpan are NamedTuple with:
|
||||
start: index of the first token
|
||||
end: index of the token following the last token
|
||||
"""
|
||||
|
||||
if not self._encodings:
|
||||
@@ -316,18 +264,15 @@ class BatchEncoding(UserDict):
|
||||
return TokenSpan(*(self._encodings[batch_index].word_to_tokens(word_index)))
|
||||
|
||||
def token_to_chars(self, batch_or_token_index: int, token_index: Optional[int] = None) -> CharSpan:
|
||||
"""
|
||||
Get the character span corresponding to an encoded token in a sequence of the batch.
|
||||
""" Get the character span corresponding to an encoded token in a sequence of the batch.
|
||||
|
||||
Character spans are returned as a CharSpan NamedTuple with:
|
||||
Character spans are returned as a CharSpan NamedTuple with:
|
||||
start: index of the first character in the original string associated to the token
|
||||
end: index of the character following the last character in the original string associated to the token
|
||||
|
||||
- start: index of the first character in the original string associated to the token
|
||||
- end: index of the character following the last character in the original string associated to the token
|
||||
|
||||
Can be called as:
|
||||
|
||||
- ``self.token_to_chars(token_index)`` if batch size is 1
|
||||
- ``self.token_to_chars(batch_index, token_index)`` if batch size is greater or equal to 1
|
||||
Can be called as:
|
||||
- self.token_to_chars(token_index) if batch size is 1
|
||||
- self.token_to_chars(batch_index, token_index) if batch size is greater or equal to 1
|
||||
|
||||
Args:
|
||||
batch_or_token_index (:obj:`int`):
|
||||
@@ -338,13 +283,12 @@ class BatchEncoding(UserDict):
|
||||
of the token or tokens in the sequence.
|
||||
|
||||
Returns:
|
||||
:obj:`CharSpan`:
|
||||
char_span (:obj:`CharSpan`):
|
||||
Span of characters in the original string.
|
||||
|
||||
:obj:`CharSpan` are NamedTuple with:
|
||||
|
||||
- start: index of the first character in the original string
|
||||
- end: index of the character following the last character in the original string
|
||||
CharSpan are NamedTuple with:
|
||||
start: index of the first character in the original string
|
||||
end: index of the character following the last character in the original string
|
||||
"""
|
||||
|
||||
if not self._encodings:
|
||||
@@ -357,18 +301,16 @@ class BatchEncoding(UserDict):
|
||||
return CharSpan(*(self._encodings[batch_index].token_to_chars(token_index)))
|
||||
|
||||
def char_to_token(self, batch_or_char_index: int, char_index: Optional[int] = None) -> int:
|
||||
"""
|
||||
Get the index of the token in the encoded output comprising a character
|
||||
in the original string for a sequence of the batch.
|
||||
""" Get the index of the token in the encoded output comprising a character
|
||||
in the original string for a sequence of the batch.
|
||||
|
||||
Can be called as:
|
||||
Can be called as:
|
||||
- self.char_to_token(char_index) if batch size is 1
|
||||
- self.char_to_token(batch_index, char_index) if batch size is greater or equal to 1
|
||||
|
||||
- ``self.char_to_token(char_index)`` if batch size is 1
|
||||
- ``self.char_to_token(batch_index, char_index)`` if batch size is greater or equal to 1
|
||||
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
|
||||
Args:
|
||||
batch_or_char_index (:obj:`int`):
|
||||
@@ -380,7 +322,8 @@ class BatchEncoding(UserDict):
|
||||
|
||||
|
||||
Returns:
|
||||
:obj:`int`: Index of the token.
|
||||
token_index (:obj:`int`):
|
||||
Index of the token.
|
||||
"""
|
||||
|
||||
if not self._encodings:
|
||||
@@ -393,19 +336,16 @@ class BatchEncoding(UserDict):
|
||||
return self._encodings[batch_index].char_to_token(char_index)
|
||||
|
||||
def word_to_chars(self, batch_or_word_index: int, word_index: Optional[int] = None) -> CharSpan:
|
||||
"""
|
||||
Get the character span in the original string corresponding to given word in a sequence
|
||||
of the batch.
|
||||
""" Get the character span in the original string corresponding to given word in a sequence
|
||||
of the batch.
|
||||
|
||||
Character spans are returned as a CharSpan NamedTuple with:
|
||||
Character spans are returned as a CharSpan NamedTuple with:
|
||||
start: index of the first character in the original string
|
||||
end: index of the character following the last character in the original string
|
||||
|
||||
- start: index of the first character in the original string
|
||||
- end: index of the character following the last character in the original string
|
||||
|
||||
Can be called as:
|
||||
|
||||
- ``self.word_to_chars(word_index)`` if batch size is 1
|
||||
- ``self.word_to_chars(batch_index, word_index)`` if batch size is greater or equal to 1
|
||||
Can be called as:
|
||||
- self.word_to_chars(word_index) if batch size is 1
|
||||
- self.word_to_chars(batch_index, word_index) if batch size is greater or equal to 1
|
||||
|
||||
Args:
|
||||
batch_or_word_index (:obj:`int`):
|
||||
@@ -416,12 +356,11 @@ class BatchEncoding(UserDict):
|
||||
of the word in the sequence.
|
||||
|
||||
Returns:
|
||||
:obj:`CharSpan` or :obj:`List[CharSpan]`:
|
||||
char_span (:obj:`CharSpan` or :obj:`List[CharSpan]`):
|
||||
Span(s) of the associated character or characters in the string.
|
||||
CharSpan are NamedTuple with:
|
||||
|
||||
- start: index of the first character associated to the token in the original string
|
||||
- end: index of the character following the last character associated to the token in the original string
|
||||
start: index of the first character associated to the token in the original string
|
||||
end: index of the character following the last character associated to the token in the original string
|
||||
"""
|
||||
|
||||
if not self._encodings:
|
||||
@@ -434,18 +373,16 @@ class BatchEncoding(UserDict):
|
||||
return CharSpan(*(self._encodings[batch_index].word_to_chars(word_index)))
|
||||
|
||||
def char_to_word(self, batch_or_char_index: int, char_index: Optional[int] = None) -> int:
|
||||
"""
|
||||
Get the word in the original string corresponding to a character in the original string of
|
||||
a sequence of the batch.
|
||||
""" Get the word in the original string corresponding to a character in the original string of
|
||||
a sequence of the batch.
|
||||
|
||||
Can be called as:
|
||||
Can be called as:
|
||||
- self.char_to_word(char_index) if batch size is 1
|
||||
- self.char_to_word(batch_index, char_index) if batch size is greater than 1
|
||||
|
||||
- ``self.char_to_word(char_index)`` if batch size is 1
|
||||
- ``self.char_to_word(batch_index, char_index)`` if batch size is greater than 1
|
||||
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
This method is particularly suited when the input sequences are provided as
|
||||
pre-tokenized sequences (i.e. words are defined by the user). In this case it allows
|
||||
to easily associate encoded tokens with provided tokenized words.
|
||||
|
||||
Args:
|
||||
batch_or_char_index (:obj:`int`):
|
||||
@@ -457,7 +394,7 @@ class BatchEncoding(UserDict):
|
||||
|
||||
|
||||
Returns:
|
||||
:obj:`int` or :obj:`List[int]`:
|
||||
token_index (:obj:`int` or :obj:`List[int]`):
|
||||
Index or indices of the associated encoded token(s).
|
||||
"""
|
||||
|
||||
@@ -470,51 +407,6 @@ class BatchEncoding(UserDict):
|
||||
char_index = batch_or_char_index
|
||||
return self._encodings[batch_index].char_to_word(char_index)
|
||||
|
||||
def convert_to_tensors(self, tensor_type: Union[None, str, TensorType], prepend_batch_axis: bool = False):
|
||||
if tensor_type is None:
|
||||
return self
|
||||
|
||||
# Convert to TensorType
|
||||
if not isinstance(tensor_type, TensorType):
|
||||
tensor_type = TensorType(tensor_type)
|
||||
|
||||
# Get a function reference for the correct framework
|
||||
if tensor_type == TensorType.TENSORFLOW and is_tf_available():
|
||||
as_tensor = tf.constant
|
||||
elif tensor_type == TensorType.PYTORCH and is_torch_available():
|
||||
as_tensor = torch.tensor
|
||||
elif tensor_type == TensorType.NUMPY:
|
||||
as_tensor = np.asarray
|
||||
else:
|
||||
raise ImportError(
|
||||
"Unable to convert output to tensors format {}, PyTorch or TensorFlow is not available.".format(
|
||||
tensor_type
|
||||
)
|
||||
)
|
||||
|
||||
# Do the tensor conversion in batch
|
||||
for key, value in self.items():
|
||||
try:
|
||||
if prepend_batch_axis:
|
||||
value = [value]
|
||||
|
||||
tensor = as_tensor(value)
|
||||
|
||||
# at-least2d
|
||||
if tensor.ndim > 2:
|
||||
tensor = tensor.squeeze(0)
|
||||
elif tensor.ndim < 2:
|
||||
tensor = tensor[None, :]
|
||||
|
||||
self[key] = tensor
|
||||
except: # noqa E722
|
||||
raise ValueError(
|
||||
"Unable to create tensor, you should probably activate truncation and/or padding "
|
||||
"with 'padding=True' 'truncation=True' to have batched tensors with the same length."
|
||||
)
|
||||
|
||||
return self
|
||||
|
||||
@torch_required
|
||||
def to(self, device: str):
|
||||
"""Send all values to device by calling v.to(device)"""
|
||||
@@ -556,6 +448,7 @@ class SpecialTokensMixin:
|
||||
if key in self.SPECIAL_TOKENS_ATTRIBUTES:
|
||||
if key == "additional_special_tokens":
|
||||
assert isinstance(value, (list, tuple)) and all(isinstance(t, str) for t in value)
|
||||
setattr(self, key, value)
|
||||
elif isinstance(value, AddedTokenFast):
|
||||
setattr(self, key, str(value))
|
||||
elif isinstance(value, str):
|
||||
@@ -565,7 +458,7 @@ class SpecialTokensMixin:
|
||||
"special token {} has to be either str or AddedTokenFast but got: {}".format(key, type(value))
|
||||
)
|
||||
|
||||
def add_special_tokens(self, special_tokens_dict):
|
||||
def add_special_tokens(self, special_tokens_dict: Dict[str, Union[str, List[str]]]) -> int:
|
||||
"""
|
||||
Add a dictionary of special tokens (eos, pad, cls...) to the encoder and link them
|
||||
to class attributes. If special tokens are NOT in the vocabulary, they are added
|
||||
@@ -620,9 +513,30 @@ class SpecialTokensMixin:
|
||||
|
||||
return added_tokens
|
||||
|
||||
def add_tokens(self, value):
|
||||
""" To be overriden by derived class to add a token in the vocabulary. """
|
||||
pass
|
||||
def add_tokens(self, value: Union[str, List[str]]) -> int:
|
||||
"""
|
||||
Add a list of new tokens to the tokenizer class. If the new tokens are not in the
|
||||
vocabulary, they are added to it with indices starting from length of the current vocabulary.
|
||||
|
||||
Args:
|
||||
new_tokens: string or list of string or AddedTokenFast. Each string is a token to add.
|
||||
Tokens are only added if they are not already in the vocabulary. AddedTokenFast wrap a string token to let you personnalize it's behavior (Whether this token should only match against single word, whether this token should strip all potential whitespaces on the left side, Whether this token should strip all potential whitespaces on the right side...).
|
||||
See details for AddedToken in HuggingFace tokenizers library.
|
||||
|
||||
Returns:
|
||||
Number of tokens added to the vocabulary.
|
||||
|
||||
Examples::
|
||||
|
||||
# Let's see how to increase the vocabulary of Bert model and tokenizer
|
||||
tokenizer = BertTokenizerFast.from_pretrained('bert-base-uncased')
|
||||
model = BertModel.from_pretrained('bert-base-uncased')
|
||||
|
||||
num_added_toks = tokenizer.add_tokens(['new_tok1', 'my_new-tok2'])
|
||||
print('We have added', num_added_toks, 'tokens')
|
||||
model.resize_token_embeddings(len(tokenizer)) # Notice: resize_token_embeddings expect to receive the full size of the new vocabulary, i.e. the length of the tokenizer.
|
||||
"""
|
||||
raise NotImplementedError() # Implemented in derived classes
|
||||
|
||||
def _maybe_update_backend(self, value):
|
||||
""" To be overriden by derived class if a backend tokenizer has to be updated. """
|
||||
@@ -832,8 +746,8 @@ ENCODE_KWARGS_DOCSTRING = r"""
|
||||
is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
|
||||
Set to True to indicate the input is already tokenized
|
||||
return_tensors (:obj:`str`, `optional`, defaults to :obj:`None`):
|
||||
Can be set to 'tf', 'pt' or 'np' to return respectively TensorFlow :obj:`tf.constant`,
|
||||
PyTorch :obj:`torch.Tensor` or Numpy :oj: `np.ndarray` instead of a list of python integers.
|
||||
Can be set to 'tf' or 'pt' to return respectively TensorFlow :obj:`tf.constant`
|
||||
or PyTorch :obj:`torch.Tensor` instead of a list of python integers.
|
||||
"""
|
||||
|
||||
ENCODE_PLUS_ADDITIONAL_KWARGS_DOCSTRING = r"""
|
||||
@@ -894,6 +808,18 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
|
||||
padding_side: str = "right"
|
||||
|
||||
NO_PAD_TOKEN_FOR_BATCH_MSG = (
|
||||
"No padding token is set for this model, therefore no batch can be made with uneven "
|
||||
"sequences. Set a padding token or adjust the lengths of the sequences building the "
|
||||
"batch so that every sequence is of the same length."
|
||||
)
|
||||
|
||||
UNEVEN_SEQUENCES_FOR_BATCH_MSG = (
|
||||
"The sequences building the batch are not of the same size, no tensor "
|
||||
"can be built. Set `pad_to_max_length=True` to pad the smaller sequences"
|
||||
"up to the larger sequence's length."
|
||||
)
|
||||
|
||||
def __init__(self, model_max_length=None, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
@@ -1055,8 +981,9 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
"added_tokens_file": ADDED_TOKENS_FILE,
|
||||
"special_tokens_map_file": SPECIAL_TOKENS_MAP_FILE,
|
||||
"tokenizer_config_file": TOKENIZER_CONFIG_FILE,
|
||||
"full_tokenizer_file": FULL_TOKENIZER_FILE,
|
||||
}
|
||||
# Look for the tokenizer main vocabulary files + the additional tokens files
|
||||
# Look for the tokenizer files
|
||||
for file_id, file_name in {**cls.vocab_files_names, **additional_files_names}.items():
|
||||
if os.path.isdir(pretrained_model_name_or_path):
|
||||
full_file_name = os.path.join(pretrained_model_name_or_path, file_name)
|
||||
@@ -1149,12 +1076,6 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
for args_name, file_path in resolved_vocab_files.items():
|
||||
if args_name not in init_kwargs:
|
||||
init_kwargs[args_name] = file_path
|
||||
if special_tokens_map_file is not None:
|
||||
with open(special_tokens_map_file, encoding="utf-8") as special_tokens_map_handle:
|
||||
special_tokens_map = json.load(special_tokens_map_handle)
|
||||
for key, value in special_tokens_map.items():
|
||||
if key not in init_kwargs:
|
||||
init_kwargs[key] = value
|
||||
|
||||
# Instantiate tokenizer.
|
||||
try:
|
||||
@@ -1169,18 +1090,19 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
tokenizer.init_inputs = init_inputs
|
||||
tokenizer.init_kwargs = init_kwargs
|
||||
|
||||
# update unique_added_tokens_encoder with special tokens for correct tokenization
|
||||
if hasattr(tokenizer, "unique_added_tokens_encoder"):
|
||||
tokenizer.unique_added_tokens_encoder.update(set(tokenizer.all_special_tokens))
|
||||
|
||||
# Add supplementary tokens.
|
||||
if added_tokens_file is not None:
|
||||
with open(added_tokens_file, encoding="utf-8") as added_tokens_handle:
|
||||
added_tok_encoder = json.load(added_tokens_handle)
|
||||
added_tok_decoder = {v: k for k, v in added_tok_encoder.items()}
|
||||
tokenizer.added_tokens_encoder.update(added_tok_encoder)
|
||||
tokenizer.added_tokens_decoder.update(added_tok_decoder)
|
||||
tokenizer.unique_added_tokens_encoder.update(set(tokenizer.added_tokens_encoder.keys()))
|
||||
for token, tok_index in sorted(added_tok_encoder.items(), key=lambda x: x[1]):
|
||||
assert tok_index == len(tokenizer), f"Unable to reload special tokens in tokenizer. List in not continuous, check file {added_tokens_file}."
|
||||
tokenizer.add_tokens(token)
|
||||
|
||||
# Map special tokens
|
||||
if special_tokens_map_file is not None:
|
||||
with open(special_tokens_map_file, encoding="utf-8") as special_tokens_map_handle:
|
||||
special_tokens_map = json.load(special_tokens_map_handle)
|
||||
tokenizer.add_special_tokens(special_tokens_map)
|
||||
|
||||
return tokenizer
|
||||
|
||||
@@ -1240,7 +1162,7 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
truncation: Union[bool, str] = False,
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
**kwargs
|
||||
):
|
||||
"""
|
||||
@@ -1389,7 +1311,7 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
is_pretokenized: bool = False,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_mask: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -1404,16 +1326,16 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
the mask for sequence classification and the overflowing elements if a ``max_length`` is specified.
|
||||
|
||||
Args:
|
||||
text (:obj:`str`, :obj:`List[str]`, :obj:`List[List[str]]``):
|
||||
text (:obj:`str`, :obj:`List[str]`, :obj:`List[List[str]]``:
|
||||
The sequence or batch of sequences to be encoded.
|
||||
Each sequence can be a string or a list of strings (pre-tokenized string).
|
||||
If the sequences are provided as list of strings (pretokenized), you must set `is_pretokenized=True`
|
||||
(to lift the ambiguity with a batch of sequences)
|
||||
text_pair (:obj:`str`, :obj:`List[str]`, :obj:`List[List[str]]``):
|
||||
text_pair (:obj:`str`, :obj:`List[str]`, :obj:`List[List[str]]``:
|
||||
The sequence or batch of sequences to be encoded.
|
||||
Each sequence can be a string or a list of strings (pre-tokenized string).
|
||||
If the sequences are provided as list of strings (pretokenized), you must set `is_pretokenized=True`
|
||||
(to lift the ambiguity with a batch of sequences)
|
||||
(to lift the ambiguity with a batch of sequences)
|
||||
"""
|
||||
is_batched = bool(
|
||||
(not is_pretokenized and isinstance(text, (list, tuple)))
|
||||
@@ -1471,7 +1393,7 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
is_pretokenized: bool = False,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_mask: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -1531,7 +1453,7 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
is_pretokenized: bool = False,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_mask: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -1559,7 +1481,7 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
is_pretokenized: bool = False,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_masks: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -1623,7 +1545,7 @@ class PreTrainedTokenizerBase(SpecialTokensMixin):
|
||||
max_length: Optional[int] = None,
|
||||
stride: int = 0,
|
||||
is_pretokenized: bool = False,
|
||||
return_tensors: Optional[Union[str, TensorType]] = None,
|
||||
return_tensors: Optional[str] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_masks: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
|
||||
@@ -26,6 +26,7 @@ from tokenizers import Encoding as EncodingFast
|
||||
from tokenizers.decoders import Decoder as DecoderFast
|
||||
from tokenizers.implementations import BaseTokenizer as BaseTokenizerFast
|
||||
|
||||
from .file_utils import is_tf_available, is_torch_available
|
||||
from .tokenization_utils_base import (
|
||||
BatchEncoding,
|
||||
PaddingStrategy,
|
||||
@@ -38,6 +39,11 @@ from .tokenization_utils_base import (
|
||||
)
|
||||
|
||||
|
||||
if is_tf_available():
|
||||
import tensorflow as tf
|
||||
if is_torch_available():
|
||||
import torch
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -55,19 +61,19 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
|
||||
Class attributes (overridden by derived classes):
|
||||
|
||||
- ``vocab_files_names``: a python ``dict`` with, as keys, the ``__init__`` keyword name of each vocabulary file
|
||||
required by the model, and as associated values, the filename for saving the associated file (string).
|
||||
- ``pretrained_vocab_files_map``: a python ``dict of dict`` the high-level keys
|
||||
being the ``__init__`` keyword name of each vocabulary file required by the model, the low-level being the
|
||||
`short-cut-names` (string) of the pretrained models with, as associated values, the `url` (string) to the
|
||||
associated pretrained vocabulary file.
|
||||
- ``max_model_input_sizes``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the pretrained
|
||||
models, and as associated values, the maximum length of the sequence inputs of this model, or None if the
|
||||
model has no maximum input size.
|
||||
- ``pretrained_init_configuration``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the
|
||||
pretrained models, and as associated values, a dictionnary of specific arguments to pass to the
|
||||
``__init__``method of the tokenizer class for this pretrained model when loading the tokenizer with the
|
||||
``from_pretrained()`` method.
|
||||
- ``vocab_files_names``: a python ``dict`` with, as keys, the ``__init__`` keyword name of each vocabulary file
|
||||
required by the model, and as associated values, the filename for saving the associated file (string).
|
||||
- ``pretrained_vocab_files_map``: a python ``dict of dict`` the high-level keys
|
||||
being the ``__init__`` keyword name of each vocabulary file required by the model, the low-level being the
|
||||
`short-cut-names` (string) of the pretrained models with, as associated values, the `url` (string) to the
|
||||
associated pretrained vocabulary file.
|
||||
- ``max_model_input_sizes``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the pretrained
|
||||
models, and as associated values, the maximum length of the sequence inputs of this model, or None if the
|
||||
model has no maximum input size.
|
||||
- ``pretrained_init_configuration``: a python ``dict`` with, as keys, the `short-cut-names` (string) of the
|
||||
pretrained models, and as associated values, a dictionnary of specific arguments to pass to the
|
||||
``__init__``method of the tokenizer class for this pretrained model when loading the tokenizer with the
|
||||
``from_pretrained()`` method.
|
||||
|
||||
Args:
|
||||
- ``tokenizer`` (`BaseTokenizerFast`): A Fast tokenizer from the HuggingFace tokenizer library (in low level Rust language)
|
||||
@@ -97,9 +103,6 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
- ``additional_special_tokens``: (`Optional`) list: a list of additional special tokens.
|
||||
Adding all special tokens here ensure they won't be split by the tokenization process.
|
||||
Will be associated to ``self.additional_special_tokens`` and ``self.additional_special_tokens_ids``
|
||||
|
||||
|
||||
.. automethod:: __call__
|
||||
"""
|
||||
|
||||
def __init__(self, tokenizer: BaseTokenizerFast, **kwargs):
|
||||
@@ -142,6 +145,7 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
def _convert_encoding(
|
||||
self,
|
||||
encoding: EncodingFast,
|
||||
return_tensors: Optional[bool] = None,
|
||||
return_token_type_ids: Optional[bool] = None,
|
||||
return_attention_mask: Optional[bool] = None,
|
||||
return_overflowing_tokens: bool = False,
|
||||
@@ -154,6 +158,8 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
Overflowing tokens are converted to additional examples (like batches) so the output values of
|
||||
the dict are lists (overflows) of lists (tokens).
|
||||
|
||||
If return_tensors is not None, these lists of lists are converted to 2-D tensors
|
||||
for input_ids, token_type_ids and attention_mask.
|
||||
Output shape: (overflows, sequence length)
|
||||
"""
|
||||
if return_token_type_ids is None:
|
||||
@@ -179,6 +185,24 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
if return_offsets_mapping:
|
||||
encoding_dict["offset_mapping"].append(e.offsets)
|
||||
|
||||
if return_tensors is not None:
|
||||
try:
|
||||
for key, value in encoding_dict.items():
|
||||
if return_tensors == "tf" and is_tf_available():
|
||||
encoding_dict[key] = tf.constant(value)
|
||||
elif return_tensors == "pt" and is_torch_available():
|
||||
encoding_dict[key] = torch.tensor(value)
|
||||
elif return_tensors is not None and verbose:
|
||||
logger.warning(
|
||||
"Unable to convert output to tensors format {}, "
|
||||
"PyTorch or TensorFlow is not available.".format(return_tensors)
|
||||
)
|
||||
except: # noqa E722
|
||||
raise ValueError(
|
||||
"Unable to create tensor, you should probably activate truncation and/or padding "
|
||||
"with 'padding=True' 'truncation=True' to have batched tensors with the same length."
|
||||
)
|
||||
|
||||
return encoding_dict
|
||||
|
||||
def convert_tokens_to_ids(self, tokens):
|
||||
@@ -209,43 +233,36 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
return self._tokenizer.decode(tokens, skip_special_tokens=skip_special_tokens)
|
||||
|
||||
def add_tokens(self, new_tokens: List[Union[str, AddedTokenFast]]) -> int:
|
||||
"""
|
||||
Add a list of new tokens to the tokenizer class. If the new tokens are not in the
|
||||
vocabulary, they are added to it with indices starting from length of the current vocabulary.
|
||||
|
||||
Args:
|
||||
new_tokens: string or list of string or :class:`~transformers.AddedTokenFast`. Each string is a token to add.
|
||||
Tokens are only added if they are not already in the vocabulary. AddedTokenFast wrap a string token to
|
||||
let you personnalize it's behavior (Whether this token should only match against single word, whether
|
||||
this token should strip all potential whitespaces on the left side, Whether this token should strip
|
||||
all potential whitespaces on the right side...).
|
||||
|
||||
See details for :class:`~transformers.AddedToken` in HuggingFace tokenizers library.
|
||||
|
||||
Returns:
|
||||
Number of tokens added to the vocabulary.
|
||||
|
||||
Examples::
|
||||
|
||||
# Let's see how to increase the vocabulary of Bert model and tokenizer
|
||||
tokenizer = BertTokenizerFast.from_pretrained('bert-base-uncased')
|
||||
model = BertModel.from_pretrained('bert-base-uncased')
|
||||
|
||||
num_added_toks = tokenizer.add_tokens(['new_tok1', 'my_new-tok2'])
|
||||
print('We have added', num_added_toks, 'tokens')
|
||||
model.resize_token_embeddings(len(tokenizer)) # Notice: resize_token_embeddings expect to receive the full size of the new vocabulary, i.e. the length of the tokenizer.
|
||||
"""
|
||||
if isinstance(new_tokens, str):
|
||||
if isinstance(new_tokens, (str, AddedTokenFast)):
|
||||
new_tokens = [new_tokens]
|
||||
# TODO This should be done in tokenizers to be really clean.
|
||||
# Removing for now
|
||||
# tokens = []
|
||||
# for token in new_tokens:
|
||||
# if self.init_kwargs.get("do_lower_case", False) and token not in self.all_special_tokens:
|
||||
# token = token.lower()
|
||||
# if token not in tokens:
|
||||
# tokens.append(token)
|
||||
return self._tokenizer.add_tokens(new_tokens)
|
||||
tokens = []
|
||||
for token in new_tokens:
|
||||
if self.init_kwargs.get("do_lower_case", False) and token not in self.all_special_tokens:
|
||||
token = token.lower()
|
||||
if token not in tokens:
|
||||
tokens.append(token)
|
||||
return self._tokenizer.add_tokens(tokens)
|
||||
|
||||
def add_special_tokens(self, special_tokens_dict: Dict[str, Union[str, List[str]]]) -> int:
|
||||
# Map special tokens to class attributes (self.pad_token...)
|
||||
num_added_tokens = super().add_special_tokens(special_tokens_dict)
|
||||
|
||||
# If the backend tokenizer the only specificities of special tokens are that
|
||||
# - they will never be processed by the model, and
|
||||
# - they will be removed while decoding.
|
||||
# But they are not mapped to special attributes in the backend so we can just
|
||||
# send a list.
|
||||
tokens = []
|
||||
for tok in special_tokens_dict.values():
|
||||
if isinstance(tok, str):
|
||||
tokens.append(tok)
|
||||
elif isinstance(tok, (list, tuple)):
|
||||
tokens += tok
|
||||
else:
|
||||
raise ValueError(f"Check special_tokens_dict input, {tok} should be str, list or tuple.")
|
||||
self._tokenizer.add_special_tokens(tokens)
|
||||
|
||||
return num_added_tokens
|
||||
|
||||
def num_special_tokens_to_add(self, pair: bool = False) -> int:
|
||||
return self._tokenizer.num_special_tokens_to_add(pair)
|
||||
@@ -376,6 +393,7 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
tokens = [
|
||||
self._convert_encoding(
|
||||
encoding=encoding,
|
||||
return_tensors=return_tensors,
|
||||
return_token_type_ids=return_token_type_ids,
|
||||
return_attention_mask=return_attention_mask,
|
||||
return_overflowing_tokens=return_overflowing_tokens,
|
||||
@@ -386,11 +404,24 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
for encoding in encodings
|
||||
]
|
||||
|
||||
# Convert the output to have dict[list] from list[dict]
|
||||
# Sanitize the output to have dict[list] from list[dict]
|
||||
sanitized = {}
|
||||
for key in tokens[0].keys():
|
||||
# To List[List[List[int]]] of shape (batch, overflows, sequence length)
|
||||
stack = [e for item in tokens for e in item[key]]
|
||||
try:
|
||||
if return_tensors == "tf":
|
||||
stack = tf.stack(stack, axis=0)
|
||||
elif return_tensors == "pt":
|
||||
stack = torch.stack(stack, dim=0)
|
||||
except: # noqa E722
|
||||
raise ValueError(
|
||||
"Unable to stack tensor, you should probably activate truncation and/or padding "
|
||||
"with 'padding=True' 'truncation=True' to have batched tensors with the same length."
|
||||
)
|
||||
# elif not return_tensors and len(stack) == 1:
|
||||
# stack = stack[0]
|
||||
|
||||
sanitized[key] = stack
|
||||
|
||||
# If returning overflowing tokens, we need to return a mapping
|
||||
@@ -401,7 +432,7 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
overflow_to_sample_mapping += [i] * len(enc["input_ids"])
|
||||
sanitized["overflow_to_sample_mapping"] = overflow_to_sample_mapping
|
||||
|
||||
return BatchEncoding(sanitized, encodings, tensor_type=return_tensors)
|
||||
return BatchEncoding(sanitized, encodings)
|
||||
|
||||
def _encode_plus(
|
||||
self,
|
||||
@@ -444,7 +475,7 @@ class PreTrainedTokenizerFast(PreTrainedTokenizerBase):
|
||||
|
||||
# Return tensor is None, then we can remove the leading batch axis
|
||||
# Overfolwing tokens are returned as a batch of output so we keep them in this case
|
||||
if return_tensors is None and not return_overflowing_tokens:
|
||||
if not return_tensors and not return_overflowing_tokens:
|
||||
batched_output = BatchEncoding(
|
||||
{
|
||||
key: value[0] if len(value) > 0 and isinstance(value[0], list) else value
|
||||
|
||||
@@ -20,10 +20,10 @@ import re
|
||||
import shutil
|
||||
import tempfile
|
||||
from collections import OrderedDict
|
||||
from typing import TYPE_CHECKING, Dict, Tuple, Union
|
||||
from typing import TYPE_CHECKING, Dict, Tuple, Union, List
|
||||
|
||||
from tests.utils import require_tf, require_torch
|
||||
from transformers import PreTrainedTokenizer, PreTrainedTokenizerFast
|
||||
from transformers import PreTrainedTokenizer, PreTrainedTokenizerFast, PreTrainedTokenizerBase
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -93,7 +93,7 @@ class TokenizerTesterMixin:
|
||||
output_ids = tokenizer.encode(output_txt, add_special_tokens=False)
|
||||
return output_txt, output_ids
|
||||
|
||||
def get_tokenizers(self, fast=True, **kwargs) -> PreTrainedTokenizer:
|
||||
def get_tokenizers(self, fast=True, **kwargs) -> List[PreTrainedTokenizerBase]:
|
||||
if fast and self.test_rust_tokenizer:
|
||||
return [self.get_tokenizer(**kwargs), self.get_rust_tokenizer(**kwargs)]
|
||||
return [self.get_tokenizer(**kwargs)]
|
||||
@@ -101,7 +101,7 @@ class TokenizerTesterMixin:
|
||||
def get_tokenizer(self, **kwargs) -> PreTrainedTokenizer:
|
||||
return self.tokenizer_class.from_pretrained(self.tmpdirname, **kwargs)
|
||||
|
||||
def get_rust_tokenizer(self, **kwargs):
|
||||
def get_rust_tokenizer(self, **kwargs) -> PreTrainedTokenizerFast:
|
||||
raise NotImplementedError
|
||||
|
||||
# def get_input_output_texts(self) -> Tuple[str, str]:
|
||||
@@ -156,25 +156,34 @@ class TokenizerTesterMixin:
|
||||
|
||||
def test_save_and_load_tokenizer(self):
|
||||
# safety check on max_len default value so we are sure the test works
|
||||
tokenizers = self.get_tokenizers(fast=False)
|
||||
tokenizers = self.get_tokenizers()
|
||||
for tokenizer in tokenizers:
|
||||
with self.subTest(f"{tokenizer.__class__.__name__}"):
|
||||
self.assertNotEqual(tokenizer.max_len, 42)
|
||||
|
||||
# Now let's start the test
|
||||
tokenizers = self.get_tokenizers(fast=False, model_max_length=42)
|
||||
tokenizers = self.get_tokenizers(model_max_length=42)
|
||||
for tokenizer in tokenizers:
|
||||
with self.subTest(f"{tokenizer.__class__.__name__}"):
|
||||
sample_text = "He is very happy, UNwant\u00E9d,running"
|
||||
tokenizer.add_tokens(["bim", "bambam"])
|
||||
additional_special_tokens = tokenizer.additional_special_tokens
|
||||
additional_special_tokens.append("new_additional_special_token")
|
||||
tokenizer.add_special_tokens({'additional_special_tokens': additional_special_tokens})
|
||||
before_tokens = tokenizer.encode(sample_text, add_special_tokens=False)
|
||||
|
||||
before_vocab = tokenizer.get_vocab()
|
||||
tokenizer.save_pretrained(self.tmpdirname)
|
||||
|
||||
tokenizer = self.tokenizer_class.from_pretrained(self.tmpdirname)
|
||||
|
||||
after_tokens = tokenizer.encode(sample_text, add_special_tokens=False)
|
||||
after_vocab = tokenizer.get_vocab()
|
||||
self.assertListEqual(before_tokens, after_tokens)
|
||||
|
||||
self.assertDictEqual(before_vocab, after_vocab)
|
||||
self.assertIn("bim", after_vocab)
|
||||
self.assertIn("bambam", after_vocab)
|
||||
self.assertIn("new_additional_special_token", tokenizer.additional_special_tokens)
|
||||
self.assertEqual(tokenizer.model_max_length, 42)
|
||||
|
||||
tokenizer = self.tokenizer_class.from_pretrained(self.tmpdirname, model_max_length=43)
|
||||
self.assertEqual(tokenizer.model_max_length, 43)
|
||||
|
||||
@@ -1297,46 +1306,10 @@ class TokenizerTesterMixin:
|
||||
model(encoded_sequence)
|
||||
model(batch_encoded_sequence)
|
||||
|
||||
# TODO: Check if require_torch is the best to test for numpy here ... Maybe move to require_flax when available
|
||||
@require_torch
|
||||
def test_np_encode_plus_sent_to_model(self):
|
||||
from transformers import MODEL_MAPPING, TOKENIZER_MAPPING
|
||||
|
||||
MODEL_TOKENIZER_MAPPING = merge_model_tokenizer_mappings(MODEL_MAPPING, TOKENIZER_MAPPING)
|
||||
|
||||
tokenizer = self.get_tokenizer()
|
||||
if tokenizer.__class__ not in MODEL_TOKENIZER_MAPPING:
|
||||
return
|
||||
|
||||
config_class, model_class = MODEL_TOKENIZER_MAPPING[tokenizer.__class__]
|
||||
config = config_class()
|
||||
|
||||
if config.is_encoder_decoder or config.pad_token_id is None:
|
||||
return
|
||||
|
||||
# Build sequence
|
||||
first_ten_tokens = list(tokenizer.get_vocab().keys())[:10]
|
||||
sequence = " ".join(first_ten_tokens)
|
||||
encoded_sequence = tokenizer.encode_plus(sequence, return_tensors="np")
|
||||
batch_encoded_sequence = tokenizer.batch_encode_plus([sequence, sequence], return_tensors="np")
|
||||
|
||||
# TODO: add forward through JAX/Flax when PR is merged
|
||||
# This is currently here to make flake8 happy !
|
||||
if encoded_sequence is None:
|
||||
raise ValueError("Cannot convert list to numpy tensor on encode_plus()")
|
||||
|
||||
if batch_encoded_sequence is None:
|
||||
raise ValueError("Cannot convert list to numpy tensor on batch_encode_plus()")
|
||||
|
||||
if self.test_rust_tokenizer:
|
||||
fast_tokenizer = self.get_rust_tokenizer()
|
||||
encoded_sequence_fast = fast_tokenizer.encode_plus(sequence, return_tensors="np")
|
||||
batch_encoded_sequence_fast = fast_tokenizer.batch_encode_plus([sequence, sequence], return_tensors="np")
|
||||
|
||||
# TODO: add forward through JAX/Flax when PR is merged
|
||||
# This is currently here to make flake8 happy !
|
||||
if encoded_sequence_fast is None:
|
||||
raise ValueError("Cannot convert list to numpy tensor on encode_plus() (fast)")
|
||||
|
||||
if batch_encoded_sequence_fast is None:
|
||||
raise ValueError("Cannot convert list to numpy tensor on batch_encode_plus() (fast)")
|
||||
# if self.test_rust_tokenizer:
|
||||
# fast_tokenizer = self.get_rust_tokenizer()
|
||||
# encoded_sequence_fast = fast_tokenizer.encode_plus(sequence, return_tensors="tf")
|
||||
# batch_encoded_sequence_fast = fast_tokenizer.batch_encode_plus([sequence, sequence], return_tensors="tf")
|
||||
# # This should not fail
|
||||
# model(encoded_sequence_fast)
|
||||
# model(batch_encoded_sequence_fast)
|
||||
|
||||
@@ -76,9 +76,6 @@ class CommonFastTokenizerTest(unittest.TestCase):
|
||||
self.assert_embeded_special_tokens(tokenizer_r, tokenizer_p)
|
||||
self.assert_padding(tokenizer_r, tokenizer_p)
|
||||
self.assert_pretokenized_inputs(tokenizer_r, tokenizer_p)
|
||||
self.assert_create_token_type_ids(tokenizer_r, tokenizer_p)
|
||||
# TODO: enable for v3.0.0
|
||||
# self.assert_empty_output_no_special_tokens(tokenizer_r, tokenizer_p)
|
||||
|
||||
def fast_only(self, tokenizer_r):
|
||||
# Ensure None raise an error
|
||||
@@ -227,7 +224,6 @@ class CommonFastTokenizerTest(unittest.TestCase):
|
||||
self.assertEqual(len(tokenizer_r), vocab_size + 3)
|
||||
|
||||
self.assertEqual(tokenizer_r.add_special_tokens({}), 0)
|
||||
self.assertEqual(tokenizer_r.add_special_tokens({"bos_token": "[BOS]", "eos_token": "[EOS]"}), 2)
|
||||
self.assertRaises(
|
||||
AssertionError, tokenizer_r.add_special_tokens, {"additional_special_tokens": "<testtoken1>"}
|
||||
)
|
||||
@@ -235,7 +231,7 @@ class CommonFastTokenizerTest(unittest.TestCase):
|
||||
self.assertEqual(
|
||||
tokenizer_r.add_special_tokens({"additional_special_tokens": ["<testtoken3>", "<testtoken4>"]}), 2
|
||||
)
|
||||
self.assertEqual(len(tokenizer_r), vocab_size + 8)
|
||||
self.assertEqual(len(tokenizer_r), vocab_size + 6)
|
||||
|
||||
def assert_offsets_mapping(self, tokenizer_r):
|
||||
text = "Wonderful no inspiration example with subtoken"
|
||||
@@ -375,20 +371,6 @@ class CommonFastTokenizerTest(unittest.TestCase):
|
||||
for key in output_p.keys():
|
||||
self.assertEqual(output_p[key], output_r[key])
|
||||
|
||||
def assert_create_token_type_ids(self, tokenizer_r, tokenizer_p):
|
||||
input_simple = [1, 2, 3]
|
||||
input_pair = [1, 2, 3]
|
||||
|
||||
# Generate output
|
||||
output_r = tokenizer_r.create_token_type_ids_from_sequences(input_simple)
|
||||
output_p = tokenizer_p.create_token_type_ids_from_sequences(input_simple)
|
||||
self.assertEqual(output_p, output_r)
|
||||
|
||||
# Generate pair output
|
||||
output_r = tokenizer_r.create_token_type_ids_from_sequences(input_simple, input_pair)
|
||||
output_p = tokenizer_p.create_token_type_ids_from_sequences(input_simple, input_pair)
|
||||
self.assertEqual(output_p, output_r)
|
||||
|
||||
def assert_build_inputs_with_special_tokens(self, tokenizer_r, tokenizer_p):
|
||||
# Input string
|
||||
input_simple = tokenizer_p.tokenize("This is a sample input")
|
||||
|
||||
Reference in New Issue
Block a user