Compare commits

...
Author SHA1 Message Date
Patrick von Platen 8f2a74bd3d fix timing 2020-07-08 09:34:18 +00:00
Patrick von Platen 884652ad18 fix timing 2020-07-08 09:26:23 +00:00
Patrick von Platen 8adb1b57f0 fix timing 2020-07-08 09:20:40 +00:00
Patrick von Platen c58d75f8a1 adapt timing for tpu 2020-07-08 09:16:05 +00:00
Patrick von Platen 9567fe2aee tf_train 2020-07-08 10:34:47 +02:00
2 changed files with 78 additions and 12 deletions
+3 -3
View File
@@ -157,7 +157,7 @@ class PyTorchBenchmark(Benchmark):
else: else:
train_model = model train_model = model
model.eval() model.train()
model.to(self.args.device) model.to(self.args.device)
# encoder-decoder has vocab size saved differently # encoder-decoder has vocab size saved differently
@@ -175,12 +175,12 @@ class PyTorchBenchmark(Benchmark):
def compute_loss_and_backprob_encoder(): def compute_loss_and_backprob_encoder():
loss = train_model(input_ids, labels=input_ids)[0] loss = train_model(input_ids, labels=input_ids)[0]
loss.backward() loss.backward()
train_model.zero_grad() return loss
def compute_loss_and_backprob_encoder_decoder(): def compute_loss_and_backprob_encoder_decoder():
loss = train_model(input_ids, decoder_input_ids=input_ids, labels=input_ids)[0] loss = train_model(input_ids, decoder_input_ids=input_ids, labels=input_ids)[0]
loss.backward() loss.backward()
train_model.zero_grad() return loss
_train = ( _train = (
compute_loss_and_backprob_encoder_decoder compute_loss_and_backprob_encoder_decoder
+75 -9
View File
@@ -21,10 +21,17 @@
import logging import logging
import random import random
import timeit import timeit
import time
from functools import wraps from functools import wraps
from typing import Callable, Optional from typing import Callable, Optional
from transformers import TF_MODEL_MAPPING, PretrainedConfig, is_py3nvml_available, is_tf_available from transformers import (
TF_MODEL_MAPPING,
TF_MODEL_WITH_LM_HEAD_MAPPING,
PretrainedConfig,
is_py3nvml_available,
is_tf_available,
)
from .benchmark_utils import ( from .benchmark_utils import (
Benchmark, Benchmark,
@@ -92,10 +99,11 @@ class TensorFlowBenchmark(Benchmark):
_inference = self._prepare_inference_func(model_name, batch_size, sequence_length) _inference = self._prepare_inference_func(model_name, batch_size, sequence_length)
return self._measure_speed(_inference) return self._measure_speed(_inference)
def _train_speed(self, model_name, batch_size, sequence_length): def _train_speed(self, model_name: str, batch_size: int, sequence_length: int) -> float:
raise NotImplementedError( strategy = self.args.strategy
"Training is currently not really implemented." "Wait for TFTrainer to support CLM and MLM." assert strategy is not None, "A device strategy has to be initialized before using TensorFlow."
) _train = self._prepare_train_func(model_name, batch_size, sequence_length)
return self._measure_speed(_train)
def _inference_memory( def _inference_memory(
self, model_name: str, batch_size: int, sequence_length: int self, model_name: str, batch_size: int, sequence_length: int
@@ -108,10 +116,16 @@ class TensorFlowBenchmark(Benchmark):
_inference = self._prepare_inference_func(model_name, batch_size, sequence_length) _inference = self._prepare_inference_func(model_name, batch_size, sequence_length)
return self._measure_memory(_inference) return self._measure_memory(_inference)
def _train_memory(self, model_name, batch_size, sequence_length): def _train_memory(
raise NotImplementedError( self, model_name: str, batch_size: int, sequence_length: int
"Training is currently not really implemented. Wait for TFTrainer to support CLM and MLM." ) -> [Memory, Optional[MemorySummary]]:
) if self.args.is_gpu:
tf.config.experimental.set_memory_growth(self.args.gpu_list[self.args.device_idx], True)
strategy = self.args.strategy
assert strategy is not None, "A device strategy has to be initialized before using TensorFlow."
_train = self._prepare_train_func(model_name, batch_size, sequence_length)
return self._measure_memory(_train)
def _prepare_inference_func(self, model_name: str, batch_size: int, sequence_length: int) -> Callable[[], None]: def _prepare_inference_func(self, model_name: str, batch_size: int, sequence_length: int) -> Callable[[], None]:
config = self.config_dict[model_name] config = self.config_dict[model_name]
@@ -149,16 +163,68 @@ class TensorFlowBenchmark(Benchmark):
return _inference return _inference
def _prepare_train_func(self, model_name: str, batch_size: int, sequence_length: int) -> Callable[[], None]:
config = self.config_dict[model_name]
assert (
self.args.eager_mode is False
), "Training cannot be done in eager mode. Please make sure that `args.eager_mode = False`."
if self.args.fp16:
raise NotImplementedError("Mixed precision is currently not supported.")
has_model_class_in_config = hasattr(config, "architecture") and len(config.architectures) > 1
if not self.args.only_pretrain_model and has_model_class_in_config:
try:
model_class = "TF" + config.architectures[0] # prepend 'TF' for tensorflow model
transformers_module = __import__("transformers", fromlist=[model_class])
model_cls = getattr(transformers_module, model_class)
model = model_cls(config)
except ImportError:
raise ImportError(
f"{model_class} does not exist. If you just want to test the pretrained model, you might want to set `--only_pretrain_model` or `args.only_pretrain_model=True`."
)
else:
model = TF_MODEL_WITH_LM_HEAD_MAPPING[config.__class__](config)
# encoder-decoder has vocab size saved differently
vocab_size = config.vocab_size if hasattr(config, "vocab_size") else config.encoder.vocab_size
input_ids = random_input_ids(batch_size, sequence_length, vocab_size)
@run_with_tf_optimizations(self.args.eager_mode, self.args.use_xla)
def encoder_decoder_train():
loss = model(input_ids, decoder_input_ids=input_ids, labels=input_ids, training=True)[0]
gradients = tf.gradients(loss, model.trainable_variables)
return gradients
@run_with_tf_optimizations(self.args.eager_mode, self.args.use_xla)
def encoder_train():
loss = model(input_ids, labels=input_ids, training=True)[0]
gradients = tf.gradients(loss, model.trainable_variables)
return gradients
_train = encoder_decoder_train if config.is_encoder_decoder else encoder_train
return _train
def _measure_speed(self, func) -> float: def _measure_speed(self, func) -> float:
with self.args.strategy.scope(): with self.args.strategy.scope():
try: try:
if self.args.is_tpu or self.args.use_xla: if self.args.is_tpu or self.args.use_xla:
# run additional 10 times to stabilize compilation for tpu # run additional 10 times to stabilize compilation for tpu
logger.info("Do inference on TPU. Running model 5 times to stabilize compilation") logger.info("Do inference on TPU. Running model 5 times to stabilize compilation")
# grads = [func() for i in range(5)]
timeit.repeat(func, repeat=1, number=5) timeit.repeat(func, repeat=1, number=5)
# as written in https://docs.python.org/2/library/timeit.html#timeit.Timer.repeat, min should be taken rather than the average # as written in https://docs.python.org/2/library/timeit.html#timeit.Timer.repeat, min should be taken rather than the average
runtimes = timeit.repeat(func, repeat=self.args.repeat, number=10,) runtimes = timeit.repeat(func, repeat=self.args.repeat, number=10,)
# start_time = time.time()
# grads = [func() for i in range(10)]
# end_time = time.time() - start_time
#
# print("Time", end_time / 10)
# print("Grads", grads[0][0])
# return end_time / 10
return min(runtimes) / 10.0 return min(runtimes) / 10.0
except ResourceExhaustedError as e: except ResourceExhaustedError as e: