codekingpro/portable-devtools
114k
1# -------------------------------------------------------------------------
2# Copyright (c) Microsoft Corporation. All rights reserved.
3# Licensed under the MIT License. See License.txt in the project root for
4# license information.
5# --------------------------------------------------------------------------
6import argparse
7import datetime
8import json
9import logging
10import os
11import subprocess
12
13import torch
14from benchmark_helper import setup_logger
15from metrics import BenchmarkRecord
16
17logger = logging.getLogger(__name__)
18
19
20def get_args():
21 parser = argparse.ArgumentParser()
22
23 parser.add_argument(
24 "-b",
25 "--batch-sizes",
26 type=str,
27 default="1 2",
28 )
29
30 parser.add_argument(
31 "-s",
32 "--sequence-lengths",
33 type=str,
34 default="8 16 32 64 128 256 512",
35 )
36
37 parser.add_argument(
38 "-w",
39 "--warmup-runs",
40 type=int,
41 default=5,
42 )
43
44 parser.add_argument(
45 "-n",
46 "--num-runs",
47 type=int,
48 default=1000,
49 )
50
51 parser.add_argument(
52 "--hf-pt-eager",
53 default=False,
54 action="store_true",
55 help="Benchmark in PyTorch without `torch.compile`",
56 )
57
58 parser.add_argument(
59 "--hf-pt-compile",
60 default=False,
61 action="store_true",
62 help="Benchmark in PyTorch with `torch.compile`",
63 )
64
65 parser.add_argument(
66 "--hf-ort-dir-path",
67 type=str,
68 default="",
69 help="Path to folder containing ONNX models for Optimum + ORT benchmarking",
70 )
71
72 parser.add_argument(
73 "--ort-msft-model-path",
74 type=str,
75 default="",
76 help="Path to ONNX model from https://github.com/microsoft/Llama-2-Onnx",
77 )
78
79 parser.add_argument(
80 "--ort-convert-to-onnx-model-path",
81 type=str,
82 default="",
83 help="Path to ONNX model from convert_to_onnx",
84 )
85
86 parser.add_argument(
87 "--cache-dir",
88 type=str,
89 default="./model_cache",
90 help="Cache dir where Hugging Face files are stored",
91 )
92
93 parser.add_argument(
94 "--model-name",
95 type=str,
96 required=True,
97 help="Model name in Hugging Face",
98 )
99
100 parser.add_argument(
101 "--precision",
102 type=str,
103 required=True,
104 choices=["int4", "int8", "fp16", "fp32"],
105 help="Precision to run model",
106 )
107
108 parser.add_argument(
109 "--device",
110 type=str,
111 required=True,
112 choices=["cpu", "cuda"],
113 help="Device to benchmark models",
114 )
115
116 parser.add_argument(
117 "--device-id",
118 type=int,
119 default=0,
120 help="GPU device ID",
121 )
122
123 parser.add_argument(
124 "--verbose",
125 default=False,
126 action="store_true",
127 help="Print detailed logs",
128 )
129
130 parser.add_argument(
131 "--timeout",
132 type=int,
133 default=10,
134 help="Number of mins to attempt the benchmark before moving on",
135 )
136
137 parser.add_argument(
138 "--log-folder",
139 type=str,
140 default=None,
141 help="Path to folder to save logs and results",
142 )
143
144 args = parser.parse_args()
145
146 setattr(args, "model_size", args.model_name.split("/")[-1].replace(".", "-")) # noqa: B010
147 log_folder_name = f"./{args.model_size}_{args.precision}"
148 if not args.log_folder:
149 args.log_folder = log_folder_name
150 os.makedirs(args.log_folder, exist_ok=True)
151
152 # Convert timeout value to secs
153 args.timeout *= 60
154
155 return args
156
157
158def process_log_file(device_id, log_file, base_results):
159 entries = []
160 batch_size, sequence_length, step = None, None, None
161 latency_s, latency_ms, throughput, memory = None, None, None, None
162
163 batch_pattern = "Batch Size: "
164 sequence_pattern = "Sequence Length: "
165 prompt_step_pattern = "to get past_key_values"
166 per_token_step_pattern = "with past_key_values"
167 latency_pattern = "Latency: "
168 throughput_pattern = "Throughput: "
169 memory_pattern = "peak="
170
171 with open(log_file) as f:
172 for input_line in f:
173 line = input_line.replace("\n", "")
174
175 if batch_pattern in line:
176 batch_size = int(line[len(batch_pattern) :])
177 elif sequence_pattern in line:
178 sequence_length = int(line[len(sequence_pattern) :])
179 elif prompt_step_pattern in line:
180 step = "prompt"
181 elif per_token_step_pattern in line:
182 step = "per-token"
183 elif latency_pattern in line:
184 latency_s = float(line[len(latency_pattern) : line.rfind(" ")])
185 latency_ms = latency_s * 1000
186 elif throughput_pattern in line:
187 throughput = float(line[len(throughput_pattern) : line.rfind(" ")])
188 elif memory_pattern in line:
189 if "CPU" in line:
190 # Example format for log entry:
191 # CPU memory usage: before=1000.0 MB, peak=2000.0 MB
192 memory = float(line[line.rfind("=") + 1 : line.rfind(" MB")]) / 1000
193 else:
194 # Example format for log entry:
195 # GPU memory usage: before=[{'device_id': 0, 'name': 'NVIDIA A100-SXM4-80GB', 'max_used_MB': 69637.25}, {'device_id': 1, 'name': 'NVIDIA A100-SXM4-80GB', 'max_used_MB': 890.625}] peak=[{'device_id': 0, 'name': 'NVIDIA A100-SXM4-80GB', 'max_used_MB': 73861.25}, {'device_id': 1, 'name': 'NVIDIA A100-SXM4-80GB', 'max_used_MB': 890.625}]
196 peak = line[line.find(memory_pattern) + len(memory_pattern) :].replace("'", '"')
197 usage = json.loads(peak)[device_id]["max_used_MB"]
198 memory = float(usage) / 1000
199
200 # Append log entry to list of entries
201 entry = base_results + [ # noqa: RUF005
202 batch_size,
203 sequence_length,
204 step,
205 latency_s,
206 latency_ms,
207 throughput,
208 memory,
209 ]
210 entries.append(entry)
211
212 return entries
213
214
215def save_results(results, filename):
216 import pandas as pd # noqa: PLC0415
217
218 df = pd.DataFrame(
219 results,
220 columns=[
221 "Warmup Runs",
222 "Measured Runs",
223 "Model Name",
224 "Engine",
225 "Precision",
226 "Device",
227 "Batch Size",
228 "Sequence Length",
229 "Step",
230 "Latency (s)",
231 "Latency (ms)",
232 "Throughput (tps)",
233 "Memory (GB)",
234 ],
235 )
236
237 # Set column types
238 df["Warmup Runs"] = df["Warmup Runs"].astype("int")
239 df["Measured Runs"] = df["Measured Runs"].astype("int")
240 df["Batch Size"] = df["Batch Size"].astype("int")
241 df["Sequence Length"] = df["Sequence Length"].astype("int")
242 df["Latency (s)"] = df["Latency (s)"].astype("float")
243 df["Latency (ms)"] = df["Latency (ms)"].astype("float")
244 df["Throughput (tps)"] = df["Throughput (tps)"].astype("float")
245 df["Memory (GB)"] = df["Memory (GB)"].astype("float")
246
247 # get package name and version
248 import pkg_resources # noqa: PLC0415
249
250 installed_packages = pkg_resources.working_set
251 installed_packages_list = sorted(
252 [f"{i.key}=={i.version}" for i in installed_packages if i.key in ["onnxruntime", "onnxruntime-gpu"]]
253 )
254
255 ort_pkg_name = ""
256 ort_pkg_version = ""
257 if installed_packages_list:
258 ort_pkg_name = installed_packages_list[0].split("==")[0]
259 ort_pkg_version = installed_packages_list[0].split("==")[1]
260
261 # Save results to csv with standard format
262 records = []
263 for _, row in df.iterrows():
264 if row["Engine"] in ["optimum-ort", "onnxruntime"]:
265 record = BenchmarkRecord(
266 row["Model Name"], row["Precision"], "onnxruntime", row["Device"], ort_pkg_name, ort_pkg_version
267 )
268 elif row["Engine"] in ["pytorch-eager", "pytorch-compile"]:
269 record = BenchmarkRecord(
270 row["Model Name"], row["Precision"], "pytorch", row["Device"], torch.__name__, torch.__version__
271 )
272 else:
273 record = BenchmarkRecord(row["Model Name"], row["Precision"], row["Engine"], row["Device"], "", "")
274 record.config.warmup_runs = row["Warmup Runs"]
275 record.config.measured_runs = row["Measured Runs"]
276 record.config.batch_size = row["Batch Size"]
277 record.config.seq_length = row["Sequence Length"]
278 record.config.customized["measure_step"] = row["Step"]
279 record.config.customized["engine"] = row["Engine"]
280 record.metrics.customized["latency_s_mean"] = row["Latency (s)"]
281 record.metrics.latency_ms_mean = row["Latency (ms)"]
282 record.metrics.customized["throughput_tps"] = row["Throughput (tps)"]
283 record.metrics.max_memory_usage_GB = row["Memory (GB)"]
284
285 records.append(record)
286
287 BenchmarkRecord.save_as_csv(filename, records)
288 BenchmarkRecord.save_as_json(filename.replace(".csv", ".json"), records)
289 logger.info(f"Results saved in {filename}!")
290
291
292def benchmark(args, benchmark_cmd, engine):
293 log_filename = f"{engine}_{datetime.datetime.now():%Y-%m-%d_%H:%M:%S}.log"
294 log_path = os.path.join(args.log_folder, log_filename)
295 with open(log_path, "w") as log_file:
296 process = subprocess.Popen(benchmark_cmd, stdout=log_file, stderr=log_file)
297 try:
298 process.wait(args.timeout)
299 except subprocess.TimeoutExpired:
300 process.kill()
301
302 # Create entries for csv
303 logger.info("Gathering data from log files...")
304 base_results = [args.warmup_runs, args.num_runs, args.model_name, engine, args.precision, args.device]
305 results = process_log_file(args.device_id, log_path, base_results)
306
307 return results
308
309
310def main():
311 args = get_args()
312 setup_logger(args.verbose)
313 logger.info(args.__dict__)
314 torch.backends.cudnn.benchmark = True
315
316 all_results = []
317 os.environ["CUDA_VISIBLE_DEVICES"] = str(args.device_id)
318
319 # Benchmark PyTorch without torch.compile
320 if args.hf_pt_eager:
321 benchmark_cmd = [
322 "python",
323 "-m",
324 "models.llama.benchmark",
325 "--benchmark-type",
326 "hf-pt-eager",
327 "--model-name",
328 args.model_name,
329 "--precision",
330 args.precision,
331 "--batch-sizes",
332 args.batch_sizes,
333 "--sequence-lengths",
334 args.sequence_lengths,
335 "--device",
336 args.device,
337 "--warmup-runs",
338 str(args.warmup_runs),
339 "--num-runs",
340 str(args.num_runs),
341 "--log-folder",
342 args.log_folder,
343 "--cache-dir",
344 args.cache_dir,
345 "--auth",
346 ]
347 logger.info("Benchmark PyTorch without torch.compile")
348 results = benchmark(args, benchmark_cmd, "pytorch-eager")
349 all_results.extend(results)
350
351 # Benchmark PyTorch with torch.compile
352 if args.hf_pt_compile:
353 benchmark_cmd = [
354 "python",
355 "-m",
356 "models.llama.benchmark",
357 "--benchmark-type",
358 "hf-pt-compile",
359 "--model-name",
360 args.model_name,
361 "--precision",
362 args.precision,
363 "--batch-sizes",
364 args.batch_sizes,
365 "--sequence-lengths",
366 args.sequence_lengths,
367 "--device",
368 args.device,
369 "--warmup-runs",
370 str(args.warmup_runs),
371 "--num-runs",
372 str(args.num_runs),
373 "--log-folder",
374 args.log_folder,
375 "--cache-dir",
376 args.cache_dir,
377 "--auth",
378 ]
379 logger.info("Benchmark PyTorch with torch.compile")
380 results = benchmark(args, benchmark_cmd, "pytorch-compile")
381 all_results.extend(results)
382
383 # Benchmark Optimum + ONNX Runtime
384 if args.hf_ort_dir_path:
385 benchmark_cmd = [
386 "python",
387 "-m",
388 "models.llama.benchmark",
389 "--benchmark-type",
390 "hf-ort",
391 "--hf-ort-dir-path",
392 args.hf_ort_dir_path,
393 "--model-name",
394 args.model_name,
395 "--precision",
396 args.precision,
397 "--batch-sizes",
398 args.batch_sizes,
399 "--sequence-lengths",
400 args.sequence_lengths,
401 "--device",
402 args.device,
403 "--warmup-runs",
404 str(args.warmup_runs),
405 "--num-runs",
406 str(args.num_runs),
407 "--log-folder",
408 args.log_folder,
409 "--cache-dir",
410 args.cache_dir,
411 "--auth",
412 ]
413 logger.info("Benchmark Optimum + ONNX Runtime")
414 results = benchmark(args, benchmark_cmd, "optimum-ort")
415 all_results.extend(results)
416
417 # Benchmark Microsoft model in ONNX Runtime
418 if args.ort_msft_model_path:
419 benchmark_cmd = [
420 "python",
421 "-m",
422 "models.llama.benchmark",
423 "--benchmark-type",
424 "ort-msft",
425 "--ort-model-path",
426 args.ort_msft_model_path,
427 "--model-name",
428 args.model_name,
429 "--precision",
430 args.precision,
431 "--batch-sizes",
432 args.batch_sizes,
433 "--sequence-lengths",
434 args.sequence_lengths,
435 "--device",
436 args.device,
437 "--warmup-runs",
438 str(args.warmup_runs),
439 "--num-runs",
440 str(args.num_runs),
441 "--log-folder",
442 args.log_folder,
443 "--cache-dir",
444 args.cache_dir,
445 ]
446 logger.info("Benchmark Microsoft model in ONNX Runtime")
447 results = benchmark(args, benchmark_cmd, "ort-msft")
448 all_results.extend(results)
449
450 # Benchmark convert_to_onnx model in ONNX Runtime
451 if args.ort_convert_to_onnx_model_path:
452 benchmark_cmd = [
453 "python",
454 "-m",
455 "models.llama.benchmark",
456 "--benchmark-type",
457 "ort-convert-to-onnx",
458 "--ort-model-path",
459 args.ort_convert_to_onnx_model_path,
460 "--model-name",
461 args.model_name,
462 "--precision",
463 args.precision,
464 "--batch-sizes",
465 args.batch_sizes,
466 "--sequence-lengths",
467 args.sequence_lengths,
468 "--device",
469 args.device,
470 "--warmup-runs",
471 str(args.warmup_runs),
472 "--num-runs",
473 str(args.num_runs),
474 "--log-folder",
475 args.log_folder,
476 "--cache-dir",
477 args.cache_dir,
478 ]
479 logger.info("Benchmark convert_to_onnx model in ONNX Runtime")
480 results = benchmark(args, benchmark_cmd, "onnxruntime")
481 all_results.extend(results)
482
483 csv_file = f"{args.model_size}_{args.precision}_{datetime.datetime.now():%Y-%m-%d_%H:%M:%S}.csv"
484 save_results(all_results, os.path.join(args.log_folder, csv_file))
485
486
487if __name__ == "__main__":
488 main()
489 