445 KiB
445 KiB
In [ ]:
#@title Check available memory of GPU
# Check that we are using 100% of GPU
# memory footprint support libraries/code
!ln -sf /opt/bin/nvidia-smi /usr/bin/nvidia-smi
!pip -q install gputil
!pip -q install psutil
!pip -q install humanize
import psutil
import humanize
import os
import GPUtil as GPU
GPUs = GPU.getGPUs()
# XXX: only one GPU on Colab and isn’t guaranteed
gpu = GPUs[0]
def printm():
process = psutil.Process(os.getpid())
print("Gen RAM Free: " + humanize.naturalsize( psutil.virtual_memory().available ), " | Proc size: " + humanize.naturalsize( process.memory_info().rss))
print("GPU RAM Free: {0:.0f}MB | Used: {1:.0f}MB | Util {2:3.0f}% | Total {3:.0f}MB".format(gpu.memoryFree, gpu.memoryUsed, gpu.memoryUtil*100, gpu.memoryTotal))
printm()Building wheel for gputil (setup.py) ... [?25l[?25hdone Gen RAM Free: 12.8 GB | Proc size: 160.0 MB GPU RAM Free: 16280MB | Used: 0MB | Util 0% | Total 16280MB
In [ ]:
# If GPU RAM Util > 0% => crash notebook on purpose
# !kill -9 -1In [ ]:
# install transformes
!pip uninstall -y transformers
!pip install -q git+https://github.com/huggingface/transformers.git
# install py3nvml to track GPU memory usage
!pip install -q py3nvml
!rm -f run_benchmark.py
!rm -f run_benchmark_tf.py
!rm -f plot_csv_file.py
!wget https://raw.githubusercontent.com/huggingface/transformers/master/examples/benchmarking/run_benchmark.py -qq
!wget https://raw.githubusercontent.com/huggingface/transformers/master/examples/benchmarking/run_benchmark_tf.py -qq
!wget https://raw.githubusercontent.com/huggingface/transformers/master/examples/benchmarking/plot_csv_file.py -qq
# import pandas to pretty print csv files
import pandas as pdIn [ ]:
!python run_benchmark.py --help2020-06-26 11:51:47.129203: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
usage: run_benchmark.py [-h] [--models MODELS [MODELS ...]]
[--batch_sizes BATCH_SIZES [BATCH_SIZES ...]]
[--sequence_lengths SEQUENCE_LENGTHS [SEQUENCE_LENGTHS ...]]
[--no_inference] [--no_cuda] [--no_tpu] [--fp16]
[--training] [--verbose] [--no_speed] [--no_memory]
[--trace_memory_line_by_line] [--save_to_csv]
[--log_print] [--no_env_print] [--no_multi_process]
[--with_lm_head]
[--inference_time_csv_file INFERENCE_TIME_CSV_FILE]
[--inference_memory_csv_file INFERENCE_MEMORY_CSV_FILE]
[--train_time_csv_file TRAIN_TIME_CSV_FILE]
[--train_memory_csv_file TRAIN_MEMORY_CSV_FILE]
[--env_info_csv_file ENV_INFO_CSV_FILE]
[--log_filename LOG_FILENAME] [--repeat REPEAT]
[--only_pretrain_model] [--torchscript]
[--torch_xla_tpu_print_metrics]
[--fp16_opt_level FP16_OPT_LEVEL]
optional arguments:
-h, --help show this help message and exit
--models MODELS [MODELS ...]
Model checkpoints to be provided to the AutoModel
classes. Leave blank to benchmark the base version of
all available models
--batch_sizes BATCH_SIZES [BATCH_SIZES ...]
List of batch sizes for which memory and time
performance will be evaluated
--sequence_lengths SEQUENCE_LENGTHS [SEQUENCE_LENGTHS ...]
List of sequence lengths for which memory and time
performance will be evaluated
--no_inference Don't benchmark inference of model
--no_cuda Whether to run on available cuda devices
--no_tpu Whether to run on available tpu devices
--fp16 Use FP16 to accelerate inference.
--training Benchmark training of model
--verbose Verbose memory tracing
--no_speed Don't perform speed measurments
--no_memory Don't perform memory measurments
--trace_memory_line_by_line
Trace memory line by line
--save_to_csv Save result to a CSV file
--log_print Save all print statements in a log file
--no_env_print Don't print environment information
--no_multi_process Don't use multiprocessing for memory and speed
measurement. It is highly recommended to use
multiprocessing for accurate CPU and GPU memory
measurements. This option should only be used for
debugging / testing and on TPU.
--with_lm_head Use model with its language model head
(MODEL_WITH_LM_HEAD_MAPPING instead of MODEL_MAPPING)
--inference_time_csv_file INFERENCE_TIME_CSV_FILE
CSV filename used if saving time results to csv.
--inference_memory_csv_file INFERENCE_MEMORY_CSV_FILE
CSV filename used if saving memory results to csv.
--train_time_csv_file TRAIN_TIME_CSV_FILE
CSV filename used if saving time results to csv for
training.
--train_memory_csv_file TRAIN_MEMORY_CSV_FILE
CSV filename used if saving memory results to csv for
training.
--env_info_csv_file ENV_INFO_CSV_FILE
CSV filename used if saving environment information.
--log_filename LOG_FILENAME
Log filename used if print statements are saved in
log.
--repeat REPEAT Times an experiment will be run.
--only_pretrain_model
Instead of loading the model as defined in
`config.architectures` if exists, just load the
pretrain model weights.
--torchscript Trace the models using torchscript
--torch_xla_tpu_print_metrics
Print Xla/PyTorch tpu metrics
--fp16_opt_level FP16_OPT_LEVEL
For fp16: Apex AMP optimization level selected in
['O0', 'O1', 'O2', and 'O3'].See details at
https://nvidia.github.io/apex/amp.html
In [ ]:
# create plots folder in content
!mkdir -p plots_ptIn [ ]:
# run benchmark
!python run_benchmark.py --no_speed --save_to_csv \
--models a-ware/roberta-large-squad-classification \
a-ware/xlmroberta-squadv2 \
aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200616_squad2 \
deepset/roberta-base-squad2 \
mrm8488/longformer-base-4096-finetuned-squadv2 \
--sequence_lengths 32 128 512 1024 \
--batch_sizes 32 \
--inference_memory_csv_file plots_pt/required_memory.csv \
--env_info_csv_file plots_pt/env.csv >/dev/null 2>&1 # redirect all printsIn [ ]:
df = pd.read_csv('plots_pt/required_memory.csv')
df| model | batch_size | sequence_length | result | |
|---|---|---|---|---|
| 0 | a-ware/roberta-large-squad-classification | 32 | 32 | 2219.0 |
| 1 | a-ware/roberta-large-squad-classification | 32 | 128 | 2455.0 |
| 2 | a-ware/roberta-large-squad-classification | 32 | 512 | 3641.0 |
| 3 | a-ware/roberta-large-squad-classification | 32 | 1024 | NaN |
| 4 | a-ware/xlmroberta-squadv2 | 32 | 32 | 2999.0 |
| 5 | a-ware/xlmroberta-squadv2 | 32 | 128 | 3235.0 |
| 6 | a-ware/xlmroberta-squadv2 | 32 | 512 | 4421.0 |
| 7 | a-ware/xlmroberta-squadv2 | 32 | 1024 | NaN |
| 8 | aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200... | 32 | 32 | 1025.0 |
| 9 | aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200... | 32 | 128 | 1143.0 |
| 10 | aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200... | 32 | 512 | 1719.0 |
| 11 | aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200... | 32 | 1024 | NaN |
| 12 | deepset/roberta-base-squad2 | 32 | 32 | 1373.0 |
| 13 | deepset/roberta-base-squad2 | 32 | 128 | 1533.0 |
| 14 | deepset/roberta-base-squad2 | 32 | 512 | 2433.0 |
| 15 | deepset/roberta-base-squad2 | 32 | 1024 | NaN |
| 16 | mrm8488/longformer-base-4096-finetuned-squadv2 | 32 | 32 | 3783.0 |
| 17 | mrm8488/longformer-base-4096-finetuned-squadv2 | 32 | 128 | 3783.0 |
| 18 | mrm8488/longformer-base-4096-finetuned-squadv2 | 32 | 512 | 3783.0 |
| 19 | mrm8488/longformer-base-4096-finetuned-squadv2 | 32 | 1024 | 6427.0 |
In [ ]:
df = pd.read_csv('plots_pt/env.csv')
df| transformers_version | 2.11.0 | |
|---|---|---|
| 0 | framework | PyTorch |
| 1 | use_torchscript | False |
| 2 | framework_version | 1.5.1+cu101 |
| 3 | python_version | 3.6.9 |
| 4 | system | Linux |
| 5 | cpu | x86_64 |
| 6 | architecture | 64bit |
| 7 | date | 2020-06-26 |
| 8 | time | 11:56:37.277009 |
| 9 | fp16 | False |
| 10 | use_multiprocessing | True |
| 11 | only_pretrain_model | False |
| 12 | cpu_ram_mb | 13021 |
| 13 | use_gpu | True |
| 14 | num_gpus | 1 |
| 15 | gpu | Tesla P100-PCIE-16GB |
| 16 | gpu_ram_mb | 16280 |
| 17 | gpu_power_watts | 250.0 |
| 18 | gpu_performance_state | 0 |
| 19 | use_tpu | False |
In [ ]:
# plot graph and save as image
!python plot_csv_file.py --csv_file plots_pt/required_memory.csv --figure_png_file=plots_pt/required_memory_plot.png --no_log_scale --short_model_names a-ware-roberta a-aware-xlm aodiniz-bert deepset-roberta mrm8488-long
# show image
from IPython.display import Image
Image('plots_pt/required_memory_plot.png')2020-06-26 11:56:39.671579: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
In [ ]:
!python run_benchmark.py --no_speed --save_to_csv \
--inference_memory_csv_file plots_pt/required_memory_2.csv \
--env_info_csv_file plots_pt/env.csv \
--models aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200616_squad2 \
deepset/roberta-base-squad2 \
--sequence_lengths 512 \
--batch_sizes 64 128 256 512\
--no_env_print2020-06-26 11:56:44.781155: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
1 / 2
2 / 2
Doesn't fit on GPU. CUDA out of memory. Tried to allocate 6.00 GiB (GPU 0; 15.90 GiB total capacity; 9.47 GiB already allocated; 5.60 GiB free; 9.52 GiB reserved in total by PyTorch)
==================== INFERENCE - MEMORY - RESULT ====================
--------------------------------------------------------------------------------
Model Name Batch Size Seq Length Memory in MB
--------------------------------------------------------------------------------
aodiniz/bert_uncased_L-10_H-51 64 512 2455
aodiniz/bert_uncased_L-10_H-51 128 512 3929
aodiniz/bert_uncased_L-10_H-51 256 512 6875
aodiniz/bert_uncased_L-10_H-51 512 512 12783
deepset/roberta-base-squad2 64 512 3539
deepset/roberta-base-squad2 128 512 5747
deepset/roberta-base-squad2 256 512 10167
deepset/roberta-base-squad2 512 512 N/A
--------------------------------------------------------------------------------
Saving results to csv.
In [ ]:
# plot graph and save as image
!python plot_csv_file.py --csv_file plots_pt/required_memory_2.csv \
--figure_png_file=plots_pt/required_memory_plot_2.png \
--no_log_scale \
--short_model_names aodiniz-bert deepset-roberta \
--plot_along_batch
# show image
from IPython.display import Image
Image('plots_pt/required_memory_plot_2.png')2020-06-26 11:57:51.876810: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
In [ ]:
# create plots folder in content
!mkdir -p plots_tf
!TF_CPP_MIN_LOG_LEVEL=3 python run_benchmark_tf.py --no_speed --save_to_csv \
--inference_memory_csv_file plots_tf/required_memory_2.csv \
--env_info_csv_file plots_tf/env.csv \
--models aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200616_squad2 \
deepset/roberta-base-squad2 \
--sequence_lengths 512 \
--batch_sizes 64 128 256 512 \
--no_env_print \1 / 2
Doesn't fit on GPU. OOM when allocating tensor with shape[512,8,512,512] and type float on /job:localhost/replica:0/task:0/device:GPU:0 by allocator GPU_0_bfc
[[node tf_bert_model/bert/encoder/layer_._0/attention/self/Softmax (defined at /usr/local/lib/python3.6/dist-packages/transformers/modeling_tf_bert.py:267) ]]
Hint: If you want to see a list of allocated tensors when OOM happens, add report_tensor_allocations_upon_oom to RunOptions for current allocation info.
[Op:__inference_run_in_graph_mode_4243]
Errors may have originated from an input operation.
Input Source operations connected to node tf_bert_model/bert/encoder/layer_._0/attention/self/Softmax:
tf_bert_model/bert/encoder/layer_._0/attention/self/add (defined at /usr/local/lib/python3.6/dist-packages/transformers/modeling_tf_bert.py:264)
Function call stack:
run_in_graph_mode
2 / 2
Doesn't fit on GPU. OOM when allocating tensor with shape[512,12,512,512] and type float on /job:localhost/replica:0/task:0/device:GPU:0 by allocator GPU_0_bfc
[[node tf_roberta_model/roberta/encoder/layer_._0/attention/self/Softmax (defined at /usr/local/lib/python3.6/dist-packages/transformers/modeling_tf_bert.py:267) ]]
Hint: If you want to see a list of allocated tensors when OOM happens, add report_tensor_allocations_upon_oom to RunOptions for current allocation info.
[Op:__inference_run_in_graph_mode_5047]
Errors may have originated from an input operation.
Input Source operations connected to node tf_roberta_model/roberta/encoder/layer_._0/attention/self/Softmax:
tf_roberta_model/roberta/encoder/layer_._0/attention/self/add (defined at /usr/local/lib/python3.6/dist-packages/transformers/modeling_tf_bert.py:264)
Function call stack:
run_in_graph_mode
==================== INFERENCE - MEMORY - RESULT ====================
--------------------------------------------------------------------------------
Model Name Batch Size Seq Length Memory in MB
--------------------------------------------------------------------------------
aodiniz/bert_uncased_L-10_H-51 64 512 2885
aodiniz/bert_uncased_L-10_H-51 128 512 4933
aodiniz/bert_uncased_L-10_H-51 256 512 9029
aodiniz/bert_uncased_L-10_H-51 512 512 N/A
deepset/roberta-base-squad2 64 512 4933
deepset/roberta-base-squad2 128 512 9029
deepset/roberta-base-squad2 256 512 15391
deepset/roberta-base-squad2 512 512 N/A
--------------------------------------------------------------------------------
Saving results to csv.
In [ ]:
# plot graph and save as image
!python plot_csv_file.py --csv_file plots_tf/required_memory_2.csv --figure_png_file=plots_tf/required_memory_plot_2.png --no_log_scale --short_model_names aodiniz-bert deepset-roberta --plot_along_batch
# show image
from IPython.display import Image
Image('plots_tf/required_memory_plot_2.png')2020-06-26 11:59:28.790462: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
In [ ]:
!TF_CPP_MIN_LOG_LEVEL=3 python run_benchmark_tf.py --no_memory --save_to_csv \
--inference_time_csv_file plots_tf/time_2.csv \
--env_info_csv_file plots_tf/env.csv \
--models aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200616_squad2 \
deepset/roberta-base-squad2 \
--sequence_lengths 8 32 128 512 \
--batch_sizes 256 \
--no_env_print \1 / 2
2 / 2
==================== INFERENCE - SPEED - RESULT ====================
--------------------------------------------------------------------------------
Model Name Batch Size Seq Length Time in s
--------------------------------------------------------------------------------
aodiniz/bert_uncased_L-10_H-51 256 8 0.033
aodiniz/bert_uncased_L-10_H-51 256 32 0.119
aodiniz/bert_uncased_L-10_H-51 256 128 0.457
aodiniz/bert_uncased_L-10_H-51 256 512 2.21
deepset/roberta-base-squad2 256 8 0.064
deepset/roberta-base-squad2 256 32 0.25
deepset/roberta-base-squad2 256 128 1.01
deepset/roberta-base-squad2 256 512 4.65
--------------------------------------------------------------------------------
Saving results to csv.
In [ ]:
# plot graph and save as image
!python plot_csv_file.py --csv_file plots_tf/time_2.csv --figure_png_file=plots_tf/time_plot_2.png --no_log_scale --short_model_names aodiniz-bert deepset-roberta --is_time
# show image
from IPython.display import Image
Image('plots_tf/time_plot_2.png')2020-06-26 12:04:58.002654: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
In [ ]:
!TF_CPP_MIN_LOG_LEVEL=3 python run_benchmark_tf.py --no_memory --save_to_csv \
--inference_time_csv_file plots_tf/time_xla_1.csv \
--env_info_csv_file plots_tf/env.csv \
--models aodiniz/bert_uncased_L-10_H-512_A-8_cord19-200616_squad2 \
--sequence_lengths 512 \
--batch_sizes 8 64 256 \
--no_env_print \
--use_xla1 / 1
==================== INFERENCE - SPEED - RESULT ====================
--------------------------------------------------------------------------------
Model Name Batch Size Seq Length Time in s
--------------------------------------------------------------------------------
aodiniz/bert_uncased_L-10_H-51 8 512 0.056
aodiniz/bert_uncased_L-10_H-51 64 512 0.402
aodiniz/bert_uncased_L-10_H-51 256 512 1.591
--------------------------------------------------------------------------------
Saving results to csv.
In [ ]:
# Imports
from transformers import BartConfig, PyTorchBenchmark, PyTorchBenchmarkArgumentsIn [ ]:
BartConfig.from_pretrained("facebook/bart-large-mnli").to_diff_dict()HBox(children=(FloatProgress(value=0.0, description='Downloading', max=908.0, style=ProgressStyle(description_…
{'_num_labels': 3,
'activation_dropout': 0.0,
'activation_function': 'gelu',
'add_bias_logits': False,
'add_final_layer_norm': False,
'attention_dropout': 0.0,
'bos_token_id': 0,
'classif_dropout': 0.0,
'd_model': 1024,
'decoder_attention_heads': 16,
'decoder_ffn_dim': 4096,
'decoder_layerdrop': 0.0,
'decoder_layers': 12,
'dropout': 0.1,
'encoder_attention_heads': 16,
'encoder_ffn_dim': 4096,
'encoder_layerdrop': 0.0,
'encoder_layers': 12,
'eos_token_id': 2,
'extra_pos_embeddings': 2,
'id2label': {0: 'contradiction', 1: 'neutral', 2: 'entailment'},
'init_std': 0.02,
'is_encoder_decoder': True,
'label2id': {'contradiction': 0, 'entailment': 2, 'neutral': 1},
'max_position_embeddings': 1024,
'model_type': 'bart',
'normalize_before': False,
'normalize_embedding': True,
'num_hidden_layers': 12,
'output_past': False,
'pad_token_id': 1,
'scale_embedding': False,
'static_position_embeddings': False,
'vocab_size': 50265}In [ ]:
config_baseline = BartConfig.from_pretrained("facebook/bart-large-mnli")
config_768_hidden = BartConfig.from_pretrained("facebook/bart-large-mnli", d_model=768)
config_8_heads = BartConfig.from_pretrained("facebook/bart-large-mnli", decoder_attention_heads=8, encoder_attention_heads=8)
config_10000_vocab = BartConfig.from_pretrained("facebook/bart-large-mnli", vocab_size=10000)
config_8_layers = BartConfig.from_pretrained("facebook/bart-large-mnli", encoder_layers=8, decoder_layers=8)In [ ]:
# define args
args = PyTorchBenchmarkArguments(models=["bart-base", "bart-768-hid", "bart-8-head", "bart-10000-voc", "bart-8-lay"],
no_speed=True,
no_inference=True,
training=True,
train_memory_csv_file="plots_pt/training_mem_fp16.csv",
save_to_csv=True,
env_info_csv_file="plots_pt/env.csv",
sequence_lengths=[64, 128, 256, 512],
batch_sizes=[8],
no_env_print=True,
fp16=True) # let's train on fp16
# create benchmark
benchmark = PyTorchBenchmark(configs=[config_baseline, config_768_hidden, config_8_heads, config_10000_vocab, config_8_layers], args=args)
# run benchmark
result = benchmark.run()1 / 5
2 / 5
3 / 5
4 / 5
5 / 5
==================== TRAIN - MEMORY - RESULTS ====================
--------------------------------------------------------------------------------
Model Name Batch Size Seq Length Memory in MB
--------------------------------------------------------------------------------
bart-base 8 64 2905
bart-base 8 128 3199
bart-base 8 256 5401
bart-base 8 512 11929
bart-768-hid 8 64 2441
bart-768-hid 8 128 2891
bart-768-hid 8 256 4963
bart-768-hid 8 512 10865
bart-8-head 8 64 2869
bart-8-head 8 128 3059
bart-8-head 8 256 4825
bart-8-head 8 512 9625
bart-10000-voc 8 64 2607
bart-10000-voc 8 128 2801
bart-10000-voc 8 256 4687
bart-10000-voc 8 512 10575
bart-8-lay 8 64 2445
bart-8-lay 8 128 2591
bart-8-lay 8 256 4187
bart-8-lay 8 512 8813
--------------------------------------------------------------------------------
Saving results to csv.
In [ ]:
# plot graph and save as image
!python plot_csv_file.py --csv_file plots_pt/training_mem_fp16.csv --figure_png_file=plots_pt/training_mem_fp16.png --no_log_scale
# show image
from IPython.display import Image
Image('plots_pt/training_mem_fp16.png')2020-06-26 12:11:47.558303: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
In [ ]:
# define args
args = PyTorchBenchmarkArguments(models=["bart-8-head", "bart-8-lay"],
no_inference=True,
training=True,
no_memory=True,
train_time_csv_file="plots_pt/training_speed_fp16.csv",
save_to_csv=True,
env_info_csv_file="plots_pt/env.csv",
sequence_lengths=[32, 128, 512],
batch_sizes=[8],
no_env_print=True,
repeat=1, # to make speed measurement faster but less accurate
no_multi_process=True, # google colab has problems with multi processing
fp16=True
)
# create benchmark
benchmark = PyTorchBenchmark(configs=[config_8_heads, config_8_layers], args=args)
# run benchmark
result = benchmark.run()1 / 2
2 / 2
==================== TRAIN - SPEED - RESULTS ====================
--------------------------------------------------------------------------------
Model Name Batch Size Seq Length Time in s
--------------------------------------------------------------------------------
bart-8-head 8 32 0.127
bart-8-head 8 128 0.398
bart-8-head 8 512 1.567
bart-8-lay 8 32 0.088
bart-8-lay 8 128 0.284
bart-8-lay 8 512 1.153
--------------------------------------------------------------------------------
Saving results to csv.
In [ ]:
# plot graph and save as image
!python plot_csv_file.py --csv_file plots_pt/training_speed_fp16.csv --figure_png_file=plots_pt/training_speed_fp16.png --no_log_scale --is_time
# show image
from IPython.display import Image
Image('plots_pt/training_speed_fp16.png')2020-06-26 12:13:17.849561: I tensorflow/stream_executor/platform/default/dso_loader.cc:44] Successfully opened dynamic library libcudart.so.10.1
