Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
bert_test_data.py642 linesDownload Raw Back to transformers
1# -------------------------------------------------------------------------
2# Copyright (c) Microsoft Corporation.  All rights reserved.
3# Licensed under the MIT License.
4# --------------------------------------------------------------------------
5
6# It is a tool to generate test data for a bert model.
7# The test data can be used by onnxruntime_perf_test tool to evaluate the inference latency.
8
9import argparse
10import os
11import random
12from pathlib import Path
13
14import numpy as np
15from onnx import ModelProto, TensorProto, numpy_helper
16from onnx_model import OnnxModel
17
18
19def fake_input_ids_data(
20    input_ids: TensorProto, batch_size: int, sequence_length: int, dictionary_size: int
21) -> np.ndarray:
22    """Create input tensor based on the graph input of input_ids
23
24    Args:
25        input_ids (TensorProto): graph input of the input_ids input tensor
26        batch_size (int): batch size
27        sequence_length (int): sequence length
28        dictionary_size (int): vocabulary size of dictionary
29
30    Returns:
31        np.ndarray: the input tensor created
32    """
33    assert input_ids.type.tensor_type.elem_type in [
34        TensorProto.FLOAT,
35        TensorProto.INT32,
36        TensorProto.INT64,
37    ]
38
39    data = np.random.randint(dictionary_size, size=(batch_size, sequence_length), dtype=np.int32)
40
41    if input_ids.type.tensor_type.elem_type == TensorProto.FLOAT:
42        data = np.float32(data)
43    elif input_ids.type.tensor_type.elem_type == TensorProto.INT64:
44        data = np.int64(data)
45
46    return data
47
48
49def fake_segment_ids_data(segment_ids: TensorProto, batch_size: int, sequence_length: int) -> np.ndarray:
50    """Create input tensor based on the graph input of segment_ids
51
52    Args:
53        segment_ids (TensorProto): graph input of the token_type_ids input tensor
54        batch_size (int): batch size
55        sequence_length (int): sequence length
56
57    Returns:
58        np.ndarray: the input tensor created
59    """
60    assert segment_ids.type.tensor_type.elem_type in [
61        TensorProto.FLOAT,
62        TensorProto.INT32,
63        TensorProto.INT64,
64    ]
65
66    data = np.zeros((batch_size, sequence_length), dtype=np.int32)
67
68    if segment_ids.type.tensor_type.elem_type == TensorProto.FLOAT:
69        data = np.float32(data)
70    elif segment_ids.type.tensor_type.elem_type == TensorProto.INT64:
71        data = np.int64(data)
72
73    return data
74
75
76def get_random_length(max_sequence_length: int, average_sequence_length: int):
77    assert average_sequence_length >= 1 and average_sequence_length <= max_sequence_length
78
79    # For uniform distribution, we find proper lower and upper bounds so that the average is in the middle.
80    if 2 * average_sequence_length > max_sequence_length:
81        return random.randint(2 * average_sequence_length - max_sequence_length, max_sequence_length)
82    else:
83        return random.randint(1, 2 * average_sequence_length - 1)
84
85
86def fake_input_mask_data(
87    input_mask: TensorProto,
88    batch_size: int,
89    sequence_length: int,
90    average_sequence_length: int,
91    random_sequence_length: bool,
92    mask_type: int = 2,
93) -> np.ndarray:
94    """Create input tensor based on the graph input of segment_ids.
95
96    Args:
97        input_mask (TensorProto): graph input of the attention mask input tensor
98        batch_size (int): batch size
99        sequence_length (int): sequence length
100        average_sequence_length (int): average sequence length excluding paddings
101        random_sequence_length (bool): whether use uniform random number for sequence length
102        mask_type (int): mask type - 1: mask index (sequence length excluding paddings). Shape is (batch_size).
103                                     2: 2D attention mask. Shape is (batch_size, sequence_length).
104                                     3: key len, cumulated lengths of query and key. Shape is (3 * batch_size + 2).
105
106    Returns:
107        np.ndarray: the input tensor created
108    """
109
110    assert input_mask.type.tensor_type.elem_type in [
111        TensorProto.FLOAT,
112        TensorProto.INT32,
113        TensorProto.INT64,
114    ]
115
116    if mask_type == 1:  # sequence length excluding paddings
117        data = np.ones((batch_size), dtype=np.int32)
118        if random_sequence_length:
119            for i in range(batch_size):
120                data[i] = get_random_length(sequence_length, average_sequence_length)
121        else:
122            for i in range(batch_size):
123                data[i] = average_sequence_length
124    elif mask_type == 2:  # 2D attention mask
125        data = np.zeros((batch_size, sequence_length), dtype=np.int32)
126        if random_sequence_length:
127            for i in range(batch_size):
128                actual_seq_len = get_random_length(sequence_length, average_sequence_length)
129                for j in range(actual_seq_len):
130                    data[i, j] = 1
131        else:
132            temp = np.ones((batch_size, average_sequence_length), dtype=np.int32)
133            data[: temp.shape[0], : temp.shape[1]] = temp
134    else:
135        assert mask_type == 3
136        data = np.zeros((batch_size * 3 + 2), dtype=np.int32)
137        if random_sequence_length:
138            for i in range(batch_size):
139                data[i] = get_random_length(sequence_length, average_sequence_length)
140
141            for i in range(batch_size + 1):
142                data[batch_size + i] = data[batch_size + i - 1] + data[i - 1] if i > 0 else 0
143                data[2 * batch_size + 1 + i] = data[batch_size + i - 1] + data[i - 1] if i > 0 else 0
144        else:
145            for i in range(batch_size):
146                data[i] = average_sequence_length
147            for i in range(batch_size + 1):
148                data[batch_size + i] = i * average_sequence_length
149                data[2 * batch_size + 1 + i] = i * average_sequence_length
150
151    if input_mask.type.tensor_type.elem_type == TensorProto.FLOAT:
152        data = np.float32(data)
153    elif input_mask.type.tensor_type.elem_type == TensorProto.INT64:
154        data = np.int64(data)
155
156    return data
157
158
159def output_test_data(directory: str, inputs: dict[str, np.ndarray]):
160    """Output input tensors of test data to a directory
161
162    Args:
163        directory (str): path of a directory
164        inputs (Dict[str, np.ndarray]): map from input name to value
165    """
166    if not os.path.exists(directory):
167        try:
168            os.mkdir(directory)
169        except OSError:
170            print(f"Creation of the directory {directory} failed")
171        else:
172            print(f"Successfully created the directory {directory} ")
173    else:
174        print(f"Warning: directory {directory} existed. Files will be overwritten.")
175
176    for index, (name, data) in enumerate(inputs.items()):
177        tensor = numpy_helper.from_array(data, name)
178        with open(os.path.join(directory, f"input_{index}.pb"), "wb") as file:
179            file.write(tensor.SerializeToString())
180
181
182def fake_test_data(
183    batch_size: int,
184    sequence_length: int,
185    test_cases: int,
186    dictionary_size: int,
187    verbose: bool,
188    random_seed: int,
189    input_ids: TensorProto,
190    segment_ids: TensorProto,
191    input_mask: TensorProto,
192    average_sequence_length: int,
193    random_sequence_length: bool,
194    mask_type: int,
195):
196    """Create given number of input data for testing
197
198    Args:
199        batch_size (int): batch size
200        sequence_length (int): sequence length
201        test_cases (int): number of test cases
202        dictionary_size (int): vocabulary size of dictionary for input_ids
203        verbose (bool): print more information or not
204        random_seed (int): random seed
205        input_ids (TensorProto): graph input of input IDs
206        segment_ids (TensorProto): graph input of token type IDs
207        input_mask (TensorProto): graph input of attention mask
208        average_sequence_length (int): average sequence length excluding paddings
209        random_sequence_length (bool): whether use uniform random number for sequence length
210        mask_type (int): mask type 1 is mask index; 2 is 2D mask; 3 is key len, cumulated lengths of query and key
211
212    Returns:
213        List[Dict[str,numpy.ndarray]]: list of test cases, where each test case is a dictionary
214                                       with input name as key and a tensor as value
215    """
216    assert input_ids is not None
217
218    np.random.seed(random_seed)
219    random.seed(random_seed)
220
221    all_inputs = []
222    for _test_case in range(test_cases):
223        input_1 = fake_input_ids_data(input_ids, batch_size, sequence_length, dictionary_size)
224        inputs = {input_ids.name: input_1}
225
226        if segment_ids:
227            inputs[segment_ids.name] = fake_segment_ids_data(segment_ids, batch_size, sequence_length)
228
229        if input_mask:
230            inputs[input_mask.name] = fake_input_mask_data(
231                input_mask, batch_size, sequence_length, average_sequence_length, random_sequence_length, mask_type
232            )
233
234        if verbose and len(all_inputs) == 0:
235            print("Example inputs", inputs)
236        all_inputs.append(inputs)
237    return all_inputs
238
239
240def generate_test_data(
241    batch_size: int,
242    sequence_length: int,
243    test_cases: int,
244    seed: int,
245    verbose: bool,
246    input_ids: TensorProto,
247    segment_ids: TensorProto,
248    input_mask: TensorProto,
249    average_sequence_length: int,
250    random_sequence_length: bool,
251    mask_type: int,
252    dictionary_size: int = 10000,
253):
254    """Create given number of input data for testing
255
256    Args:
257        batch_size (int): batch size
258        sequence_length (int): sequence length
259        test_cases (int): number of test cases
260        seed (int): random seed
261        verbose (bool): print more information or not
262        input_ids (TensorProto): graph input of input IDs
263        segment_ids (TensorProto): graph input of token type IDs
264        input_mask (TensorProto): graph input of attention mask
265        average_sequence_length (int): average sequence length excluding paddings
266        random_sequence_length (bool): whether use uniform random number for sequence length
267        mask_type (int): mask type 1 is mask index; 2 is 2D mask; 3 is key len, cumulated lengths of query and key
268
269    Returns:
270        List[Dict[str,numpy.ndarray]]: list of test cases, where each test case is a dictionary
271                                       with input name as key and a tensor as value
272    """
273    all_inputs = fake_test_data(
274        batch_size,
275        sequence_length,
276        test_cases,
277        dictionary_size,
278        verbose,
279        seed,
280        input_ids,
281        segment_ids,
282        input_mask,
283        average_sequence_length,
284        random_sequence_length,
285        mask_type,
286    )
287    if len(all_inputs) != test_cases:
288        print("Failed to create test data for test.")
289    return all_inputs
290
291
292def get_graph_input_from_embed_node(onnx_model, embed_node, input_index):
293    if input_index >= len(embed_node.input):
294        return None
295
296    input = embed_node.input[input_index]
297    graph_input = onnx_model.find_graph_input(input)
298    if graph_input is None:
299        parent_node = onnx_model.get_parent(embed_node, input_index)
300        if parent_node is not None and parent_node.op_type == "Cast":
301            graph_input = onnx_model.find_graph_input(parent_node.input[0])
302    return graph_input
303
304
305def find_bert_inputs(
306    onnx_model: OnnxModel,
307    input_ids_name: str | None = None,
308    segment_ids_name: str | None = None,
309    input_mask_name: str | None = None,
310) -> tuple[np.ndarray | None, np.ndarray | None, np.ndarray | None]:
311    """Find graph inputs for BERT model.
312    First, we will deduce inputs from EmbedLayerNormalization node.
313    If not found, we will guess the meaning of graph inputs based on naming.
314
315    Args:
316        onnx_model (OnnxModel): onnx model object
317        input_ids_name (str, optional): Name of graph input for input IDs. Defaults to None.
318        segment_ids_name (str, optional): Name of graph input for segment IDs. Defaults to None.
319        input_mask_name (str, optional): Name of graph input for attention mask. Defaults to None.
320
321    Raises:
322        ValueError: Graph does not have input named of input_ids_name or segment_ids_name or input_mask_name
323        ValueError: Expected graph input number does not match with specified input_ids_name, segment_ids_name
324                    and input_mask_name
325
326    Returns:
327        Tuple[Optional[np.ndarray], Optional[np.ndarray], Optional[np.ndarray]]: input tensors of input_ids,
328                                                                                 segment_ids and input_mask
329    """
330
331    graph_inputs = onnx_model.get_graph_inputs_excluding_initializers()
332
333    if input_ids_name is not None:
334        input_ids = onnx_model.find_graph_input(input_ids_name)
335        if input_ids is None:
336            raise ValueError(f"Graph does not have input named {input_ids_name}")
337
338        segment_ids = None
339        if segment_ids_name:
340            segment_ids = onnx_model.find_graph_input(segment_ids_name)
341            if segment_ids is None:
342                raise ValueError(f"Graph does not have input named {segment_ids_name}")
343
344        input_mask = None
345        if input_mask_name:
346            input_mask = onnx_model.find_graph_input(input_mask_name)
347            if input_mask is None:
348                raise ValueError(f"Graph does not have input named {input_mask_name}")
349
350        expected_inputs = 1 + (1 if segment_ids else 0) + (1 if input_mask else 0)
351        if len(graph_inputs) != expected_inputs:
352            raise ValueError(f"Expect the graph to have {expected_inputs} inputs. Got {len(graph_inputs)}")
353
354        return input_ids, segment_ids, input_mask
355
356    if len(graph_inputs) != 3:
357        raise ValueError(f"Expect the graph to have 3 inputs. Got {len(graph_inputs)}")
358
359    embed_nodes = onnx_model.get_nodes_by_op_type("EmbedLayerNormalization")
360    if len(embed_nodes) == 1:
361        embed_node = embed_nodes[0]
362        input_ids = get_graph_input_from_embed_node(onnx_model, embed_node, 0)
363        segment_ids = get_graph_input_from_embed_node(onnx_model, embed_node, 1)
364        input_mask = get_graph_input_from_embed_node(onnx_model, embed_node, 7)
365
366        if input_mask is None:
367            for input in graph_inputs:
368                input_name_lower = input.name.lower()
369                if "mask" in input_name_lower:
370                    input_mask = input
371        if input_mask is None:
372            raise ValueError("Failed to find attention mask input")
373
374        return input_ids, segment_ids, input_mask
375
376    # Try guess the inputs based on naming.
377    input_ids = None
378    segment_ids = None
379    input_mask = None
380    for input in graph_inputs:
381        input_name_lower = input.name.lower()
382        if "mask" in input_name_lower:  # matches input with name like "attention_mask" or "input_mask"
383            input_mask = input
384        elif (
385            "token" in input_name_lower or "segment" in input_name_lower
386        ):  # matches input with name like "segment_ids" or "token_type_ids"
387            segment_ids = input
388        else:
389            input_ids = input
390
391    if input_ids and segment_ids and input_mask:
392        return input_ids, segment_ids, input_mask
393
394    raise ValueError("Fail to assign 3 inputs. You might try rename the graph inputs.")
395
396
397def get_bert_inputs(
398    onnx_file: str,
399    input_ids_name: str | None = None,
400    segment_ids_name: str | None = None,
401    input_mask_name: str | None = None,
402) -> tuple[np.ndarray | None, np.ndarray | None, np.ndarray | None]:
403    """Find graph inputs for BERT model.
404    First, we will deduce inputs from EmbedLayerNormalization node.
405    If not found, we will guess the meaning of graph inputs based on naming.
406
407    Args:
408        onnx_file (str): onnx model path
409        input_ids_name (str, optional): Name of graph input for input IDs. Defaults to None.
410        segment_ids_name (str, optional): Name of graph input for segment IDs. Defaults to None.
411        input_mask_name (str, optional): Name of graph input for attention mask. Defaults to None.
412
413    Returns:
414        Tuple[Optional[np.ndarray], Optional[np.ndarray], Optional[np.ndarray]]: input tensors of input_ids,
415                                                                                 segment_ids and input_mask
416    """
417    model = ModelProto()
418    with open(onnx_file, "rb") as file:
419        model.ParseFromString(file.read())
420
421    onnx_model = OnnxModel(model)
422    return find_bert_inputs(onnx_model, input_ids_name, segment_ids_name, input_mask_name)
423
424
425def parse_arguments():
426    parser = argparse.ArgumentParser()
427
428    parser.add_argument("--model", required=True, type=str, help="bert onnx model path.")
429
430    parser.add_argument(
431        "--output_dir",
432        required=False,
433        type=str,
434        default=None,
435        help="output test data path. Default is current directory.",
436    )
437
438    parser.add_argument("--batch_size", required=False, type=int, default=1, help="batch size of input")
439
440    parser.add_argument(
441        "--sequence_length",
442        required=False,
443        type=int,
444        default=128,
445        help="maximum sequence length of input",
446    )
447
448    parser.add_argument(
449        "--input_ids_name",
450        required=False,
451        type=str,
452        default=None,
453        help="input name for input ids",
454    )
455    parser.add_argument(
456        "--segment_ids_name",
457        required=False,
458        type=str,
459        default=None,
460        help="input name for segment ids",
461    )
462    parser.add_argument(
463        "--input_mask_name",
464        required=False,
465        type=str,
466        default=None,
467        help="input name for attention mask",
468    )
469
470    parser.add_argument(
471        "--samples",
472        required=False,
473        type=int,
474        default=1,
475        help="number of test cases to be generated",
476    )
477
478    parser.add_argument("--seed", required=False, type=int, default=3, help="random seed")
479
480    parser.add_argument(
481        "--verbose",
482        required=False,
483        action="store_true",
484        help="print verbose information",
485    )
486    parser.set_defaults(verbose=False)
487
488    parser.add_argument(
489        "--only_input_tensors",
490        required=False,
491        action="store_true",
492        help="only save input tensors and no output tensors",
493    )
494    parser.set_defaults(only_input_tensors=False)
495
496    parser.add_argument(
497        "-a",
498        "--average_sequence_length",
499        default=-1,
500        type=int,
501        help="average sequence length excluding padding",
502    )
503
504    parser.add_argument(
505        "-r",
506        "--random_sequence_length",
507        required=False,
508        action="store_true",
509        help="use uniform random instead of fixed sequence length",
510    )
511    parser.set_defaults(random_sequence_length=False)
512
513    parser.add_argument(
514        "--mask_type",
515        required=False,
516        type=int,
517        default=2,
518        help="mask type: (1: mask index, 2: raw 2D mask, 3: key lengths, cumulated lengths of query and key)",
519    )
520
521    args = parser.parse_args()
522    return args
523
524
525def create_and_save_test_data(
526    model: str,
527    output_dir: str,
528    batch_size: int,
529    sequence_length: int,
530    test_cases: int,
531    seed: int,
532    verbose: bool,
533    input_ids_name: str | None,
534    segment_ids_name: str | None,
535    input_mask_name: str | None,
536    only_input_tensors: bool,
537    average_sequence_length: int,
538    random_sequence_length: bool,
539    mask_type: int,
540):
541    """Create test data for a model, and save test data to a directory.
542
543    Args:
544        model (str): path of ONNX bert model
545        output_dir (str): output directory
546        batch_size (int): batch size
547        sequence_length (int): sequence length
548        test_cases (int): number of test cases
549        seed (int): random seed
550        verbose (bool): whether print more information
551        input_ids_name (str): graph input name of input_ids
552        segment_ids_name (str): graph input name of segment_ids
553        input_mask_name (str): graph input name of input_mask
554        only_input_tensors (bool): only save input tensors,
555        average_sequence_length (int): average sequence length excluding paddings
556        random_sequence_length (bool): whether use uniform random number for sequence length
557        mask_type(int): mask type
558    """
559    input_ids, segment_ids, input_mask = get_bert_inputs(model, input_ids_name, segment_ids_name, input_mask_name)
560
561    all_inputs = generate_test_data(
562        batch_size,
563        sequence_length,
564        test_cases,
565        seed,
566        verbose,
567        input_ids,
568        segment_ids,
569        input_mask,
570        average_sequence_length,
571        random_sequence_length,
572        mask_type,
573    )
574
575    for i, inputs in enumerate(all_inputs):
576        directory = os.path.join(output_dir, "test_data_set_" + str(i))
577        output_test_data(directory, inputs)
578
579    if only_input_tensors:
580        return
581
582    import onnxruntime  # noqa: PLC0415
583
584    providers = (
585        ["CUDAExecutionProvider", "CPUExecutionProvider"]
586        if "CUDAExecutionProvider" in onnxruntime.get_available_providers()
587        else ["CPUExecutionProvider"]
588    )
589    session = onnxruntime.InferenceSession(model, providers=providers)
590    output_names = [output.name for output in session.get_outputs()]
591
592    for i, inputs in enumerate(all_inputs):
593        directory = os.path.join(output_dir, "test_data_set_" + str(i))
594        result = session.run(output_names, inputs)
595        for i, output_name in enumerate(output_names):  # noqa: PLW2901
596            tensor_result = numpy_helper.from_array(np.asarray(result[i]), output_name)
597            with open(os.path.join(directory, f"output_{i}.pb"), "wb") as file:
598                file.write(tensor_result.SerializeToString())
599
600
601def main():
602    args = parse_arguments()
603
604    if args.average_sequence_length <= 0:
605        args.average_sequence_length = args.sequence_length
606
607    output_dir = args.output_dir
608    if output_dir is None:
609        # Default output directory is a sub-directory under the directory of model.
610        p = Path(args.model)
611        output_dir = os.path.join(p.parent, f"batch_{args.batch_size}_seq_{args.sequence_length}")
612
613    if output_dir is not None:
614        # create the output directory if not existed
615        path = Path(output_dir)
616        path.mkdir(parents=True, exist_ok=True)
617    else:
618        print("Directory existed. test data files will be overwritten.")
619
620    create_and_save_test_data(
621        args.model,
622        output_dir,
623        args.batch_size,
624        args.sequence_length,
625        args.samples,
626        args.seed,
627        args.verbose,
628        args.input_ids_name,
629        args.segment_ids_name,
630        args.input_mask_name,
631        args.only_input_tensors,
632        args.average_sequence_length,
633        args.random_sequence_length,
634        args.mask_type,
635    )
636
637    print("Test data is saved to directory:", output_dir)
638
639
640if __name__ == "__main__":
641    main()
642 
codekingpro/portable-devtools · Team Ai