Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8f2a74bd3d | ||
|
|
884652ad18 | ||
|
|
8adb1b57f0 | ||
|
|
c58d75f8a1 | ||
|
|
9567fe2aee |
@@ -157,7 +157,7 @@ class PyTorchBenchmark(Benchmark):
|
|||||||
else:
|
else:
|
||||||
train_model = model
|
train_model = model
|
||||||
|
|
||||||
model.eval()
|
model.train()
|
||||||
model.to(self.args.device)
|
model.to(self.args.device)
|
||||||
|
|
||||||
# encoder-decoder has vocab size saved differently
|
# encoder-decoder has vocab size saved differently
|
||||||
@@ -175,12 +175,12 @@ class PyTorchBenchmark(Benchmark):
|
|||||||
def compute_loss_and_backprob_encoder():
|
def compute_loss_and_backprob_encoder():
|
||||||
loss = train_model(input_ids, labels=input_ids)[0]
|
loss = train_model(input_ids, labels=input_ids)[0]
|
||||||
loss.backward()
|
loss.backward()
|
||||||
train_model.zero_grad()
|
return loss
|
||||||
|
|
||||||
def compute_loss_and_backprob_encoder_decoder():
|
def compute_loss_and_backprob_encoder_decoder():
|
||||||
loss = train_model(input_ids, decoder_input_ids=input_ids, labels=input_ids)[0]
|
loss = train_model(input_ids, decoder_input_ids=input_ids, labels=input_ids)[0]
|
||||||
loss.backward()
|
loss.backward()
|
||||||
train_model.zero_grad()
|
return loss
|
||||||
|
|
||||||
_train = (
|
_train = (
|
||||||
compute_loss_and_backprob_encoder_decoder
|
compute_loss_and_backprob_encoder_decoder
|
||||||
|
|||||||
@@ -21,10 +21,17 @@
|
|||||||
import logging
|
import logging
|
||||||
import random
|
import random
|
||||||
import timeit
|
import timeit
|
||||||
|
import time
|
||||||
from functools import wraps
|
from functools import wraps
|
||||||
from typing import Callable, Optional
|
from typing import Callable, Optional
|
||||||
|
|
||||||
from transformers import TF_MODEL_MAPPING, PretrainedConfig, is_py3nvml_available, is_tf_available
|
from transformers import (
|
||||||
|
TF_MODEL_MAPPING,
|
||||||
|
TF_MODEL_WITH_LM_HEAD_MAPPING,
|
||||||
|
PretrainedConfig,
|
||||||
|
is_py3nvml_available,
|
||||||
|
is_tf_available,
|
||||||
|
)
|
||||||
|
|
||||||
from .benchmark_utils import (
|
from .benchmark_utils import (
|
||||||
Benchmark,
|
Benchmark,
|
||||||
@@ -92,10 +99,11 @@ class TensorFlowBenchmark(Benchmark):
|
|||||||
_inference = self._prepare_inference_func(model_name, batch_size, sequence_length)
|
_inference = self._prepare_inference_func(model_name, batch_size, sequence_length)
|
||||||
return self._measure_speed(_inference)
|
return self._measure_speed(_inference)
|
||||||
|
|
||||||
def _train_speed(self, model_name, batch_size, sequence_length):
|
def _train_speed(self, model_name: str, batch_size: int, sequence_length: int) -> float:
|
||||||
raise NotImplementedError(
|
strategy = self.args.strategy
|
||||||
"Training is currently not really implemented." "Wait for TFTrainer to support CLM and MLM."
|
assert strategy is not None, "A device strategy has to be initialized before using TensorFlow."
|
||||||
)
|
_train = self._prepare_train_func(model_name, batch_size, sequence_length)
|
||||||
|
return self._measure_speed(_train)
|
||||||
|
|
||||||
def _inference_memory(
|
def _inference_memory(
|
||||||
self, model_name: str, batch_size: int, sequence_length: int
|
self, model_name: str, batch_size: int, sequence_length: int
|
||||||
@@ -108,10 +116,16 @@ class TensorFlowBenchmark(Benchmark):
|
|||||||
_inference = self._prepare_inference_func(model_name, batch_size, sequence_length)
|
_inference = self._prepare_inference_func(model_name, batch_size, sequence_length)
|
||||||
return self._measure_memory(_inference)
|
return self._measure_memory(_inference)
|
||||||
|
|
||||||
def _train_memory(self, model_name, batch_size, sequence_length):
|
def _train_memory(
|
||||||
raise NotImplementedError(
|
self, model_name: str, batch_size: int, sequence_length: int
|
||||||
"Training is currently not really implemented. Wait for TFTrainer to support CLM and MLM."
|
) -> [Memory, Optional[MemorySummary]]:
|
||||||
)
|
if self.args.is_gpu:
|
||||||
|
tf.config.experimental.set_memory_growth(self.args.gpu_list[self.args.device_idx], True)
|
||||||
|
strategy = self.args.strategy
|
||||||
|
assert strategy is not None, "A device strategy has to be initialized before using TensorFlow."
|
||||||
|
|
||||||
|
_train = self._prepare_train_func(model_name, batch_size, sequence_length)
|
||||||
|
return self._measure_memory(_train)
|
||||||
|
|
||||||
def _prepare_inference_func(self, model_name: str, batch_size: int, sequence_length: int) -> Callable[[], None]:
|
def _prepare_inference_func(self, model_name: str, batch_size: int, sequence_length: int) -> Callable[[], None]:
|
||||||
config = self.config_dict[model_name]
|
config = self.config_dict[model_name]
|
||||||
@@ -149,16 +163,68 @@ class TensorFlowBenchmark(Benchmark):
|
|||||||
|
|
||||||
return _inference
|
return _inference
|
||||||
|
|
||||||
|
def _prepare_train_func(self, model_name: str, batch_size: int, sequence_length: int) -> Callable[[], None]:
|
||||||
|
config = self.config_dict[model_name]
|
||||||
|
|
||||||
|
assert (
|
||||||
|
self.args.eager_mode is False
|
||||||
|
), "Training cannot be done in eager mode. Please make sure that `args.eager_mode = False`."
|
||||||
|
|
||||||
|
if self.args.fp16:
|
||||||
|
raise NotImplementedError("Mixed precision is currently not supported.")
|
||||||
|
|
||||||
|
has_model_class_in_config = hasattr(config, "architecture") and len(config.architectures) > 1
|
||||||
|
if not self.args.only_pretrain_model and has_model_class_in_config:
|
||||||
|
try:
|
||||||
|
model_class = "TF" + config.architectures[0] # prepend 'TF' for tensorflow model
|
||||||
|
transformers_module = __import__("transformers", fromlist=[model_class])
|
||||||
|
model_cls = getattr(transformers_module, model_class)
|
||||||
|
model = model_cls(config)
|
||||||
|
except ImportError:
|
||||||
|
raise ImportError(
|
||||||
|
f"{model_class} does not exist. If you just want to test the pretrained model, you might want to set `--only_pretrain_model` or `args.only_pretrain_model=True`."
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
model = TF_MODEL_WITH_LM_HEAD_MAPPING[config.__class__](config)
|
||||||
|
|
||||||
|
# encoder-decoder has vocab size saved differently
|
||||||
|
vocab_size = config.vocab_size if hasattr(config, "vocab_size") else config.encoder.vocab_size
|
||||||
|
input_ids = random_input_ids(batch_size, sequence_length, vocab_size)
|
||||||
|
|
||||||
|
@run_with_tf_optimizations(self.args.eager_mode, self.args.use_xla)
|
||||||
|
def encoder_decoder_train():
|
||||||
|
loss = model(input_ids, decoder_input_ids=input_ids, labels=input_ids, training=True)[0]
|
||||||
|
gradients = tf.gradients(loss, model.trainable_variables)
|
||||||
|
return gradients
|
||||||
|
|
||||||
|
@run_with_tf_optimizations(self.args.eager_mode, self.args.use_xla)
|
||||||
|
def encoder_train():
|
||||||
|
loss = model(input_ids, labels=input_ids, training=True)[0]
|
||||||
|
gradients = tf.gradients(loss, model.trainable_variables)
|
||||||
|
return gradients
|
||||||
|
|
||||||
|
_train = encoder_decoder_train if config.is_encoder_decoder else encoder_train
|
||||||
|
|
||||||
|
return _train
|
||||||
|
|
||||||
def _measure_speed(self, func) -> float:
|
def _measure_speed(self, func) -> float:
|
||||||
with self.args.strategy.scope():
|
with self.args.strategy.scope():
|
||||||
try:
|
try:
|
||||||
if self.args.is_tpu or self.args.use_xla:
|
if self.args.is_tpu or self.args.use_xla:
|
||||||
# run additional 10 times to stabilize compilation for tpu
|
# run additional 10 times to stabilize compilation for tpu
|
||||||
logger.info("Do inference on TPU. Running model 5 times to stabilize compilation")
|
logger.info("Do inference on TPU. Running model 5 times to stabilize compilation")
|
||||||
|
# grads = [func() for i in range(5)]
|
||||||
timeit.repeat(func, repeat=1, number=5)
|
timeit.repeat(func, repeat=1, number=5)
|
||||||
|
|
||||||
# as written in https://docs.python.org/2/library/timeit.html#timeit.Timer.repeat, min should be taken rather than the average
|
# as written in https://docs.python.org/2/library/timeit.html#timeit.Timer.repeat, min should be taken rather than the average
|
||||||
runtimes = timeit.repeat(func, repeat=self.args.repeat, number=10,)
|
runtimes = timeit.repeat(func, repeat=self.args.repeat, number=10,)
|
||||||
|
# start_time = time.time()
|
||||||
|
# grads = [func() for i in range(10)]
|
||||||
|
# end_time = time.time() - start_time
|
||||||
|
#
|
||||||
|
# print("Time", end_time / 10)
|
||||||
|
# print("Grads", grads[0][0])
|
||||||
|
# return end_time / 10
|
||||||
|
|
||||||
return min(runtimes) / 10.0
|
return min(runtimes) / 10.0
|
||||||
except ResourceExhaustedError as e:
|
except ResourceExhaustedError as e:
|
||||||
|
|||||||
Reference in New Issue
Block a user