muellerzr/performance-debugging
0
1# coding=utf-82# Copyright 2021 The HuggingFace Inc. team. All rights reserved.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16import evaluate17import torch18from datasets import load_dataset19from torch.optim import AdamW20from torch.utils.data import DataLoader21from transformers import AutoModelForSequenceClassification, AutoTokenizer, get_linear_schedule_with_warmup22 23from accelerate import Accelerator, DistributedType24from accelerate.utils import set_seed25 26 27def get_dataloaders(accelerator: Accelerator, batch_size: int = 16):28 """29 Creates a set of `DataLoader`s for the `glue` dataset,30 using "bert-base-cased" as the tokenizer.31 32 Args:33 accelerator (`Accelerator`):34 An `Accelerator` object35 batch_size (`int`, *optional*):36 The batch size for the train and validation DataLoaders.37 """38 tokenizer = AutoTokenizer.from_pretrained("bert-base-cased")39 datasets = load_dataset("glue", "mrpc")40 41 def tokenize_function(examples):42 # max_length=None => use the model max length (it's actually the default)43 outputs = tokenizer(examples["sentence1"], examples["sentence2"], truncation=True, max_length=None)44 return outputs45 46 # Apply the method we just defined to all the examples in all the splits of the dataset47 # starting with the main process first:48 with accelerator.main_process_first():49 tokenized_datasets = datasets.map(50 tokenize_function,51 batched=True,52 remove_columns=["idx", "sentence1", "sentence2"],53 )54 55 # We also rename the 'label' column to 'labels' which is the expected name for labels by the models of the56 # transformers library57 tokenized_datasets = tokenized_datasets.rename_column("label", "labels")58 59 def collate_fn(examples):60 # On TPU it's best to pad everything to the same length or training will be very slow.61 max_length = 128 if accelerator.distributed_type == DistributedType.TPU else None62 # When using mixed precision we want round multiples of 8/1663 if accelerator.mixed_precision != "no":64 pad_to_multiple_of = 865 else:66 pad_to_multiple_of = None67 68 return tokenizer.pad(69 examples,70 padding="longest",71 max_length=max_length,72 pad_to_multiple_of=pad_to_multiple_of,73 return_tensors="pt",74 )75 76 # Instantiate dataloaders.77 train_dataloader = DataLoader(78 tokenized_datasets["train"], shuffle=True, collate_fn=collate_fn, batch_size=batch_size, drop_last=True79 )80 eval_dataloader = DataLoader(81 tokenized_datasets["validation"],82 shuffle=False,83 collate_fn=collate_fn,84 batch_size=32,85 drop_last=(accelerator.mixed_precision == "fp8"),86 )87 88 return train_dataloader, eval_dataloader89 90 91def training_function(config):92 # Initialize accelerator93 accelerator = Accelerator(94 mixed_precision="fp16",95 log_with="aim",96 project_dir="aim_logs"97 )98 # Sample hyper-parameters for learning rate, batch size, seed and a few other HPs99 lr = config["lr"]100 num_epochs = int(config["num_epochs"])101 seed = int(config["seed"])102 batch_size = 16 if accelerator.num_processes > 1 else 32103 config["batch_size"] = batch_size104 metric = evaluate.load("glue", "mrpc")105 106 set_seed(seed, device_specific=True)107 train_dataloader, eval_dataloader = get_dataloaders(accelerator, batch_size)108 model = AutoModelForSequenceClassification.from_pretrained("bert-base-cased", return_dict=True)109 lr = lr * accelerator.num_processes110 111 optimizer = AdamW(params=model.parameters(), lr=lr)112 lr_scheduler = get_linear_schedule_with_warmup(113 optimizer=optimizer,114 num_warmup_steps=0,115 num_training_steps=(len(train_dataloader) * num_epochs),116 )117 118 model, optimizer, train_dataloader, eval_dataloader, lr_scheduler = accelerator.prepare(119 model, optimizer, train_dataloader, eval_dataloader, lr_scheduler120 )121 122 accelerator.init_trackers(f'{accelerator.num_processes}_gpus', config)123 124 current_step = 0125 for epoch in range(num_epochs):126 model.train()127 total_loss = 0128 for _, batch in enumerate(train_dataloader):129 lr = lr_scheduler.get_lr()130 outputs = model(**batch)131 loss = outputs.loss132 batch_loss = accelerator.gather(loss).detach().mean().cpu().float()133 total_loss += batch_loss134 current_step += 1135 accelerator.log(136 {137 "batch_loss":batch_loss,138 "learning_rate":lr,139 }, 140 step=current_step, 141 log_kwargs={"aim":{"epoch":epoch}}142 )143 accelerator.backward(loss)144 optimizer.step()145 lr_scheduler.step()146 optimizer.zero_grad()147 current_step += 1148 149 model.eval()150 for step, batch in enumerate(eval_dataloader):151 # We could avoid this line since we set the accelerator with `device_placement=True`.152 batch.to(accelerator.device)153 with torch.no_grad():154 outputs = model(**batch)155 predictions = outputs.logits.argmax(dim=-1)156 predictions, references = accelerator.gather_for_metrics((predictions, batch["labels"]))157 metric.add_batch(158 predictions=predictions,159 references=references,160 )161 162 eval_metric = metric.compute()163 164 # Use accelerator.print to print only on the main process.165 accelerator.print(f"epoch {epoch}:", eval_metric)166 167 accelerator.log(168 {169 "accuracy": eval_metric["accuracy"],170 "f1": eval_metric["f1"],171 "train_loss": total_loss.item() / len(train_dataloader),172 },173 log_kwargs = {"aim":{"epoch":epoch}}174 )175 accelerator.end_training()176 177 178def main():179 config = {"lr": 2e-5, "num_epochs": 3, "seed": 42}180 training_function(config)181 182 183if __name__ == "__main__":184 main()185 