Arulkumar03/Wheat_HEAD_Detection_Counting_ComputerVision_Model
0
1# Copyright (c) Facebook, Inc. and its affiliates.2import logging3import numpy as np4from itertools import count5from typing import List, Tuple6import torch7import tqdm8from fvcore.common.timer import Timer9 10from detectron2.utils import comm11 12from .build import build_batch_data_loader13from .common import DatasetFromList, MapDataset14from .samplers import TrainingSampler15 16logger = logging.getLogger(__name__)17 18 19class _EmptyMapDataset(torch.utils.data.Dataset):20 """21 Map anything to emptiness.22 """23 24 def __init__(self, dataset):25 self.ds = dataset26 27 def __len__(self):28 return len(self.ds)29 30 def __getitem__(self, idx):31 _ = self.ds[idx]32 return [0]33 34 35def iter_benchmark(36 iterator, num_iter: int, warmup: int = 5, max_time_seconds: float = 6037) -> Tuple[float, List[float]]:38 """39 Benchmark an iterator/iterable for `num_iter` iterations with an extra40 `warmup` iterations of warmup.41 End early if `max_time_seconds` time is spent on iterations.42 43 Returns:44 float: average time (seconds) per iteration45 list[float]: time spent on each iteration. Sometimes useful for further analysis.46 """47 num_iter, warmup = int(num_iter), int(warmup)48 49 iterator = iter(iterator)50 for _ in range(warmup):51 next(iterator)52 timer = Timer()53 all_times = []54 for curr_iter in tqdm.trange(num_iter):55 start = timer.seconds()56 if start > max_time_seconds:57 num_iter = curr_iter58 break59 next(iterator)60 all_times.append(timer.seconds() - start)61 avg = timer.seconds() / num_iter62 return avg, all_times63 64 65class DataLoaderBenchmark:66 """67 Some common benchmarks that help understand perf bottleneck of a standard dataloader68 made of dataset, mapper and sampler.69 """70 71 def __init__(72 self,73 dataset,74 *,75 mapper,76 sampler=None,77 total_batch_size,78 num_workers=0,79 max_time_seconds: int = 90,80 ):81 """82 Args:83 max_time_seconds (int): maximum time to spent for each benchmark84 other args: same as in `build.py:build_detection_train_loader`85 """86 if isinstance(dataset, list):87 dataset = DatasetFromList(dataset, copy=False, serialize=True)88 if sampler is None:89 sampler = TrainingSampler(len(dataset))90 91 self.dataset = dataset92 self.mapper = mapper93 self.sampler = sampler94 self.total_batch_size = total_batch_size95 self.num_workers = num_workers96 self.per_gpu_batch_size = self.total_batch_size // comm.get_world_size()97 98 self.max_time_seconds = max_time_seconds99 100 def _benchmark(self, iterator, num_iter, warmup, msg=None):101 avg, all_times = iter_benchmark(iterator, num_iter, warmup, self.max_time_seconds)102 if msg is not None:103 self._log_time(msg, avg, all_times)104 return avg, all_times105 106 def _log_time(self, msg, avg, all_times, distributed=False):107 percentiles = [np.percentile(all_times, k, interpolation="nearest") for k in [1, 5, 95, 99]]108 if not distributed:109 logger.info(110 f"{msg}: avg={1.0/avg:.1f} it/s, "111 f"p1={percentiles[0]:.2g}s, p5={percentiles[1]:.2g}s, "112 f"p95={percentiles[2]:.2g}s, p99={percentiles[3]:.2g}s."113 )114 return115 avg_per_gpu = comm.all_gather(avg)116 percentiles_per_gpu = comm.all_gather(percentiles)117 if comm.get_rank() > 0:118 return119 for idx, avg, percentiles in zip(count(), avg_per_gpu, percentiles_per_gpu):120 logger.info(121 f"GPU{idx} {msg}: avg={1.0/avg:.1f} it/s, "122 f"p1={percentiles[0]:.2g}s, p5={percentiles[1]:.2g}s, "123 f"p95={percentiles[2]:.2g}s, p99={percentiles[3]:.2g}s."124 )125 126 def benchmark_dataset(self, num_iter, warmup=5):127 """128 Benchmark the speed of taking raw samples from the dataset.129 """130 131 def loader():132 while True:133 for k in self.sampler:134 yield self.dataset[k]135 136 self._benchmark(loader(), num_iter, warmup, "Dataset Alone")137 138 def benchmark_mapper(self, num_iter, warmup=5):139 """140 Benchmark the speed of taking raw samples from the dataset and map141 them in a single process.142 """143 144 def loader():145 while True:146 for k in self.sampler:147 yield self.mapper(self.dataset[k])148 149 self._benchmark(loader(), num_iter, warmup, "Single Process Mapper (sec/sample)")150 151 def benchmark_workers(self, num_iter, warmup=10):152 """153 Benchmark the dataloader by tuning num_workers to [0, 1, self.num_workers].154 """155 candidates = [0, 1]156 if self.num_workers not in candidates:157 candidates.append(self.num_workers)158 159 dataset = MapDataset(self.dataset, self.mapper)160 for n in candidates:161 loader = build_batch_data_loader(162 dataset,163 self.sampler,164 self.total_batch_size,165 num_workers=n,166 )167 self._benchmark(168 iter(loader),169 num_iter * max(n, 1),170 warmup * max(n, 1),171 f"DataLoader ({n} workers, bs={self.per_gpu_batch_size})",172 )173 del loader174 175 def benchmark_IPC(self, num_iter, warmup=10):176 """177 Benchmark the dataloader where each worker outputs nothing. This178 eliminates the IPC overhead compared to the regular dataloader.179 180 PyTorch multiprocessing's IPC only optimizes for torch tensors.181 Large numpy arrays or other data structure may incur large IPC overhead.182 """183 n = self.num_workers184 dataset = _EmptyMapDataset(MapDataset(self.dataset, self.mapper))185 loader = build_batch_data_loader(186 dataset, self.sampler, self.total_batch_size, num_workers=n187 )188 self._benchmark(189 iter(loader),190 num_iter * max(n, 1),191 warmup * max(n, 1),192 f"DataLoader ({n} workers, bs={self.per_gpu_batch_size}) w/o comm",193 )194 195 def benchmark_distributed(self, num_iter, warmup=10):196 """197 Benchmark the dataloader in each distributed worker, and log results of198 all workers. This helps understand the final performance as well as199 the variances among workers.200 201 It also prints startup time (first iter) of the dataloader.202 """203 gpu = comm.get_world_size()204 dataset = MapDataset(self.dataset, self.mapper)205 n = self.num_workers206 loader = build_batch_data_loader(207 dataset, self.sampler, self.total_batch_size, num_workers=n208 )209 210 timer = Timer()211 loader = iter(loader)212 next(loader)213 startup_time = timer.seconds()214 logger.info("Dataloader startup time: {:.2f} seconds".format(startup_time))215 216 comm.synchronize()217 218 avg, all_times = self._benchmark(loader, num_iter * max(n, 1), warmup * max(n, 1))219 del loader220 self._log_time(221 f"DataLoader ({gpu} GPUs x {n} workers, total bs={self.total_batch_size})",222 avg,223 all_times,224 True,225 )226 