Add power argument for TF PolynomialDecay (#5732)

* 🚩 Add `power` argument for TF PolynomialDecay

* 🚩 Create default optimizer with power

* 🚩 Add argument to training args

* 🚨 Clean code format

* 🚨 Fix black warning

* 🚨 Fix code format
This commit is contained in:
Cola
2020-10-05 05:16:29 -04:00
committed by GitHub
parent 41c3a3b98e
commit 60de910e60
3 changed files with 10 additions and 0 deletions
+4
View File
@@ -88,6 +88,7 @@ def create_optimizer(
adam_beta2: float = 0.999,
adam_epsilon: float = 1e-8,
weight_decay_rate: float = 0.0,
power: float = 1.0,
include_in_weight_decay: Optional[List[str]] = None,
):
"""
@@ -110,6 +111,8 @@ def create_optimizer(
The epsilon to use in Adam.
weight_decay_rate (:obj:`float`, `optional`, defaults to 0):
The weight decay to use.
power (:obj:`float`, `optional`, defaults to 1.0):
The power to use for PolynomialDecay.
include_in_weight_decay (:obj:`List[str]`, `optional`):
List of the parameter names (or re patterns) to apply weight decay to. If none is passed, weight decay is
applied to all parameters except bias and layer norm parameters.
@@ -119,6 +122,7 @@ def create_optimizer(
initial_learning_rate=init_lr,
decay_steps=num_train_steps - num_warmup_steps,
end_learning_rate=init_lr * min_lr_ratio,
power=power,
)
if num_warmup_steps:
lr_schedule = WarmUp(
+1
View File
@@ -227,6 +227,7 @@ class TFTrainer:
adam_beta2=self.args.adam_beta2,
adam_epsilon=self.args.adam_epsilon,
weight_decay_rate=self.args.weight_decay,
power=self.args.poly_power,
)
def setup_wandb(self):
+5
View File
@@ -112,6 +112,11 @@ class TFTrainingArguments(TrainingArguments):
metadata={"help": "Name of TPU"},
)
poly_power: float = field(
default=1.0,
metadata={"help": "Power for the Polynomial decay LR scheduler."},
)
xla: bool = field(default=False, metadata={"help": "Whether to activate the XLA compilation or not"})
@cached_property