Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
float16.py510 linesDownload Raw Back to transformers
1# -------------------------------------------------------------------------
2# Copyright (c) Microsoft Corporation.  All rights reserved.
3# Licensed under the MIT License.
4# --------------------------------------------------------------------------
5
6# This file is modified from https://github.com/microsoft/onnxconverter-common/blob/master/onnxconverter_common/float16.py
7# Modifications:
8# (1) Update default value of min_positive_val and max_finite_val
9# (2) keep_io_types can be list of names
10# (3) convert initializers if needed to preserve precision
11# (4) add force_fp16_initializers option
12# (5) handle Resize and GroupNorm with mixed float inputs
13# (6) allow convert_float_to_float16 to accept model path
14
15import itertools
16import logging
17import os
18import tempfile
19
20import numpy as np
21import onnx
22from onnx import AttributeProto, GraphProto, ModelProto, NodeProto, TensorProto, helper, numpy_helper
23from onnx.shape_inference import infer_shapes, infer_shapes_path
24from packaging import version
25
26logger = logging.getLogger(__name__)
27
28
29def _npfloat16_to_int(np_list):
30    """
31    Convert numpy float16 to python int.
32
33    :param np_list: numpy float16 list
34    :return int_list: python int list
35    """
36    return [int(bin(_.view("H"))[2:].zfill(16), 2) for _ in np_list]
37
38
39def convert_np_to_float16(np_array, min_positive_val=5.96e-08, max_finite_val=65504.0):
40    """
41    Convert float32 numpy array to float16 without changing sign or finiteness.
42    Positive values less than min_positive_val are mapped to min_positive_val.
43    Positive finite values greater than max_finite_val are mapped to max_finite_val.
44    Similar for negative values. NaN, 0, inf, and -inf are unchanged.
45    """
46
47    def between(a, b, c):
48        return np.logical_and(a < b, b < c)
49
50    if np_array[np.where(np_array > 0)].shape[0] > 0:
51        positive_max = np_array[np.where(np_array > 0)].max()
52        positive_min = np_array[np.where(np_array > 0)].min()
53        if positive_max >= max_finite_val:
54            logger.debug(f"the float32 number {positive_max} will be truncated to {max_finite_val}")
55        if positive_min <= min_positive_val:
56            logger.debug(f"the float32 number {positive_min} will be truncated to {min_positive_val}")
57
58    if np_array[np.where(np_array < 0)].shape[0] > 0:
59        negative_max = np_array[np.where(np_array < 0)].max()
60        negative_min = np_array[np.where(np_array < 0)].min()
61        if negative_min <= -max_finite_val:
62            logger.debug(f"the float32 number {negative_min} will be truncated to {-max_finite_val}")
63        if negative_max >= -min_positive_val:
64            logger.debug(f"the float32 number {negative_max} will be truncated to {-min_positive_val}")
65
66    np_array = np.where(between(0, np_array, min_positive_val), min_positive_val, np_array)
67    np_array = np.where(between(-min_positive_val, np_array, 0), -min_positive_val, np_array)
68    np_array = np.where(between(max_finite_val, np_array, float("inf")), max_finite_val, np_array)
69    np_array = np.where(between(float("-inf"), np_array, -max_finite_val), -max_finite_val, np_array)
70    return np.float16(np_array)
71
72
73def convert_tensor_float_to_float16(tensor, min_positive_val=5.96e-08, max_finite_val=65504.0):
74    """Convert tensor float to float16.
75
76    Args:
77        tensor (TensorProto): the tensor to convert.
78        min_positive_val (float, optional): minimal positive value. Defaults to 1e-7.
79        max_finite_val (float, optional): maximal finite value. Defaults to 1e4.
80
81    Raises:
82        ValueError: input type is not TensorProto.
83
84    Returns:
85        TensorProto: the converted tensor.
86    """
87
88    if not isinstance(tensor, TensorProto):
89        raise ValueError(f"Expected input type is an ONNX TensorProto but got {type(tensor)}")
90
91    if tensor.data_type == TensorProto.FLOAT:
92        tensor.data_type = TensorProto.FLOAT16
93        # convert float_data (float type) to float16 and write to int32_data
94        if tensor.float_data:
95            float16_data = convert_np_to_float16(np.array(tensor.float_data), min_positive_val, max_finite_val)
96            int_list = _npfloat16_to_int(float16_data)
97            tensor.int32_data[:] = int_list
98            tensor.float_data[:] = []
99        # convert raw_data (bytes type)
100        if tensor.raw_data:
101            # convert n.raw_data to float
102            float32_list = np.frombuffer(tensor.raw_data, dtype="float32")
103            # convert float to float16
104            float16_list = convert_np_to_float16(float32_list, min_positive_val, max_finite_val)
105            # convert float16 to bytes and write back to raw_data
106            tensor.raw_data = float16_list.tobytes()
107    return tensor
108
109
110def make_value_info_from_tensor(tensor):
111    shape = numpy_helper.to_array(tensor).shape
112    return helper.make_tensor_value_info(tensor.name, tensor.data_type, shape)
113
114
115DEFAULT_OP_BLOCK_LIST = [
116    "ArrayFeatureExtractor",
117    "Binarizer",
118    "CastMap",
119    "CategoryMapper",
120    "DictVectorizer",
121    "FeatureVectorizer",
122    "Imputer",
123    "LabelEncoder",
124    "LinearClassifier",
125    "LinearRegressor",
126    "Normalizer",
127    "OneHotEncoder",
128    "RandomUniformLike",
129    "SVMClassifier",
130    "SVMRegressor",
131    "Scaler",
132    "TreeEnsembleClassifier",
133    "TreeEnsembleRegressor",
134    "TreeEnsemble",
135    "ZipMap",
136    "NonMaxSuppression",
137    "TopK",
138    "RoiAlign",
139    "Range",
140    "CumSum",
141    "Min",
142    "Max",
143    "Upsample",
144]
145
146
147# Some operators has data type fixed as float for some inputs. Key is op_type, value is list of input indices
148# Note that DirectML allows float16 gamma and beta in GroupNorm. Use force_fp16_inputs parameter could overwrite this.
149ALWAYS_FLOAT_INPUTS = {"Resize": [2], "GroupNorm": [1, 2], "SkipGroupNorm": [1, 2]}
150
151
152class InitializerTracker:
153    """Class for keeping track of initializer."""
154
155    def __init__(self, initializer: TensorProto):
156        self.initializer = initializer
157        self.fp32_nodes = []
158        self.fp16_nodes = []
159
160    def add_node(self, node: NodeProto, is_node_blocked):
161        if is_node_blocked:
162            self.fp32_nodes.append(node)
163        else:
164            self.fp16_nodes.append(node)
165
166
167def convert_float_to_float16(
168    model,
169    min_positive_val=5.96e-08,
170    max_finite_val=65504.0,
171    keep_io_types=False,
172    disable_shape_infer=False,
173    op_block_list=None,
174    node_block_list=None,
175    force_fp16_initializers=False,
176    force_fp16_inputs=None,
177    use_bfloat16_as_blocked_nodes_dtype=False,
178):
179    """Convert tensor float type in the input ONNX model to tensor float16.
180
181    Args:
182        model (ModelProto or str): The ONNX model or path of the model to convert.
183        min_positive_val (float, optional): minimal positive value. Defaults to 5.96e-08.
184        max_finite_val (float, optional): maximal finite value of float16. Defaults to 65504.
185        keep_io_types (Union[bool, List[str]], optional): It could be boolean or a list of float32 input/output names.
186                                                          If True, model inputs/outputs should be left as float32.
187                                                          Defaults to False.
188        disable_shape_infer (bool, optional): Skips running onnx shape/type inference.
189                                              Useful if shape inference has been done. Defaults to False.
190        op_block_list (List[str], optional): List of op types to leave as float32.
191                                             Defaults to None, which will use `float16.DEFAULT_OP_BLOCK_LIST`.
192        node_block_list (List[str], optional): List of node names to leave as float32. Defaults to None.
193        force_fp16_initializers(bool): force converting all float initializers to float16.
194                                       Default to false, which will convert only the one needed to avoid precision loss.
195        force_fp16_inputs(Dict[str, List[int]]): Force the conversion of the inputs of some operators to float16, even if
196                                                 this script's preference it to keep them in float32.
197    Raises:
198        ValueError: input type is not ModelProto.
199
200    Returns:
201        ModelProto: converted model.
202    """
203    assert min_positive_val >= 5.96e-08, (
204        "invalid min_positive_val. smallest positive float16 value: subnormal 5.96e-08, and normalized 6.104e-05"
205    )
206    assert max_finite_val <= float(np.finfo(np.float16).max), "invalid max_finite_val. largest float16 value: 65504"
207
208    force_fp16_inputs_dict = {} if force_fp16_inputs is None else force_fp16_inputs
209
210    if isinstance(model, str):
211        model_path = model
212        if version.parse(onnx.__version__) >= version.parse("1.8.0") and not disable_shape_infer:
213            # shape_infer_model_path should be in the same folder of model_path
214            with tempfile.NamedTemporaryFile(dir=os.path.dirname(model_path)) as tmpfile:
215                shape_infer_model_path = tmpfile.name
216                # infer_shapes_path can be used for model >2GB, and infer_shapes cannot.
217                infer_shapes_path(model_path, shape_infer_model_path)
218                model = onnx.load(shape_infer_model_path)
219                disable_shape_infer = True
220        else:
221            model = onnx.load(model_path)
222
223    if not isinstance(model, ModelProto):
224        raise ValueError(f"Expected an ONNX ModelProto but got {type(model)}")
225
226    func_infer_shape = None
227    if not disable_shape_infer and version.parse(onnx.__version__) >= version.parse("1.2.0"):
228        try:
229            func_infer_shape = infer_shapes
230        finally:
231            pass
232
233    # create blocklists
234    if op_block_list is None:
235        op_block_list = DEFAULT_OP_BLOCK_LIST
236    if node_block_list is None:
237        node_block_list = []
238    op_block_list = set(op_block_list)
239    node_block_list = set(node_block_list)
240
241    # Build opset-aware always_float_inputs: Resize input layout differs between opset 10 and 11+.
242    # Opset 10: [X, scales] — scales at index 1 must stay float32.
243    # Opset 11+: [X, roi, scales, sizes] — scales at index 2 must stay float32; roi (index 1) allows fp16.
244    onnx_opset = max((o.version for o in model.opset_import if o.domain in ("", "ai.onnx")), default=11)
245    always_float_inputs = dict(ALWAYS_FLOAT_INPUTS)
246    if onnx_opset <= 10:
247        always_float_inputs["Resize"] = [1]
248
249    logger.debug(
250        f"fp16 parameters: min_positive_val={min_positive_val} max_finite_val={max_finite_val} keep_io_types={keep_io_types} disable_shape_infer={disable_shape_infer} op_block_list={op_block_list} node_block_list={node_block_list} force_fp16_initializers={force_fp16_initializers}"
251    )
252
253    # create a queue for BFS
254    queue = []
255    value_info_list = []
256    node_list = []
257
258    # Some operators (Like Resize or GroupNorm) have data type fixed as float for some input.
259    # When it is converted to float16, there are mixed types: some inputs are float32 and some are float16.
260    # This list keeps track of such nodes that are not in block list.
261    mixed_float_type_node_list = []
262
263    # type inference on input model
264    if func_infer_shape is not None:
265        model = func_infer_shape(model)
266    queue.append(model)
267    name_mapping = {}
268    graph_io_to_skip = set()
269    io_casts = set()
270
271    fp32_inputs = [n.name for n in model.graph.input if n.type.tensor_type.elem_type == TensorProto.FLOAT]
272    fp32_outputs = [n.name for n in model.graph.output if n.type.tensor_type.elem_type == TensorProto.FLOAT]
273    if isinstance(keep_io_types, list):
274        fp32_inputs = [n for n in fp32_inputs if n in keep_io_types]
275        fp32_outputs = [n for n in fp32_outputs if n in keep_io_types]
276    elif not keep_io_types:
277        fp32_inputs = []
278        fp32_outputs = []
279
280    for i, n in enumerate(model.graph.input):
281        if n.name in fp32_inputs:
282            output_name = "graph_input_cast_" + str(i)
283            name_mapping[n.name] = output_name
284            graph_io_to_skip.add(n.name)
285
286            node_name = "graph_input_cast" + str(i)
287            new_value_info = model.graph.value_info.add()
288            new_value_info.CopyFrom(n)
289            new_value_info.name = output_name
290            new_value_info.type.tensor_type.elem_type = TensorProto.FLOAT16
291            # add Cast node (from tensor(float) to tensor(float16) after graph input
292            new_node = [helper.make_node("Cast", [n.name], [output_name], to=TensorProto.FLOAT16, name=node_name)]
293            model.graph.node.extend(new_node)
294            value_info_list.append(new_value_info)
295            io_casts.add(node_name)
296
297    for i, n in enumerate(model.graph.output):
298        if n.name in fp32_outputs:
299            input_name = "graph_output_cast_" + str(i)
300            name_mapping[n.name] = input_name
301            graph_io_to_skip.add(n.name)
302
303            node_name = "graph_output_cast" + str(i)
304            # add Cast node (from tensor(float16) to tensor(float) before graph output
305            new_value_info = model.graph.value_info.add()
306            new_value_info.CopyFrom(n)
307            new_value_info.name = input_name
308            new_value_info.type.tensor_type.elem_type = TensorProto.FLOAT16
309            new_node = [helper.make_node("Cast", [input_name], [n.name], to=1, name=node_name)]
310            model.graph.node.extend(new_node)
311            value_info_list.append(new_value_info)
312            io_casts.add(node_name)
313
314    fp32_initializers: dict[str, InitializerTracker] = {}
315    while queue:
316        next_level = []
317        for q in queue:
318            # if q is model, push q.graph (GraphProto)
319            if isinstance(q, ModelProto):
320                next_level.append(q.graph)
321            # if q is model.graph, push q.node.attribute (AttributeProto)
322            if isinstance(q, GraphProto):
323                for n in q.initializer:  # TensorProto type
324                    if n.data_type == TensorProto.FLOAT:
325                        assert n.name not in fp32_initializers
326                        fp32_initializers[n.name] = InitializerTracker(n)
327
328                for n in q.node:
329                    # if n is in the block list (doesn't support float16), no conversion for the node,
330                    # and save the node for further processing
331                    if n.name in io_casts:
332                        continue
333                    for i in range(len(n.input)):
334                        if n.input[i] in name_mapping:
335                            n.input[i] = name_mapping[n.input[i]]
336                    for i in range(len(n.output)):
337                        if n.output[i] in name_mapping:
338                            n.output[i] = name_mapping[n.output[i]]
339
340                    is_node_blocked = n.op_type in op_block_list or n.name in node_block_list
341                    for i, input_name in enumerate(n.input):
342                        if input_name in fp32_initializers:
343                            # For Resize/GroupNorm, only the first input can be float16
344                            use_fp32_weight = is_node_blocked or (
345                                i in always_float_inputs.get(n.op_type, [])
346                                and i not in force_fp16_inputs_dict.get(n.op_type, [])
347                            )
348                            fp32_initializers[input_name].add_node(n, use_fp32_weight)
349
350                    if is_node_blocked:
351                        node_list.append(n)
352                    else:
353                        if n.op_type == "Cast":
354                            for attr in n.attribute:
355                                if attr.name == "to" and attr.i == TensorProto.FLOAT:
356                                    attr.i = TensorProto.FLOAT16
357                                    break
358
359                        if n.op_type in [
360                            "EyeLike",
361                            "Multinomial",
362                            "RandomNormal",
363                            "RandomNormalLike",
364                            "RandomUniform",
365                            "RandomUniformLike",
366                            "SequenceEmpty",
367                            "Bernoulli",
368                        ]:
369                            has_dtype = False
370                            for attr in n.attribute:
371                                if attr.name == "dtype":
372                                    has_dtype = True
373                                    if attr.i == TensorProto.FLOAT:
374                                        attr.i = TensorProto.FLOAT16
375
376                            # The dtype attribute is optional and default is FLOAT in the following operators
377                            # so we need add dtype attribute to specify the data type float16
378                            if (n.op_type in ["RandomNormal", "RandomUniform", "SequenceEmpty"]) and not has_dtype:
379                                n.attribute.extend([helper.make_attribute("dtype", TensorProto.FLOAT16)])
380
381                        # For Resize/GroupNorm, attribute data type cannot be changed
382                        if n.op_type not in always_float_inputs or n.op_type in force_fp16_inputs_dict:
383                            for attr in n.attribute:
384                                next_level.append(attr)  # noqa: PERF402
385                        else:
386                            mixed_float_type_node_list.append(n)
387
388            # if q is model.graph.node.attribute, push q.g and q.graphs (GraphProto)
389            # and process node.attribute.t and node.attribute.tensors (TensorProto)
390            if isinstance(q, AttributeProto):
391                next_level.append(q.g)
392                for n in q.graphs:
393                    next_level.append(n)  # noqa: PERF402
394                q.t.CopyFrom(convert_tensor_float_to_float16(q.t, min_positive_val, max_finite_val))
395                for n in q.tensors:
396                    n = convert_tensor_float_to_float16(n, min_positive_val, max_finite_val)  # noqa: PLW2901
397            # if q is graph, process input, output and value_info (ValueInfoProto)
398            if isinstance(q, GraphProto):
399                # Note that float initializers tracked by fp32_initializers will be processed later.
400                # for all ValueInfoProto with tensor(float) type in input, output and value_info, convert them to
401                # tensor(float16) except map and seq(map). And save them in value_info_list for further processing
402                for n in itertools.chain(q.input, q.output, q.value_info):
403                    if n.type.tensor_type.elem_type == TensorProto.FLOAT:
404                        if n.name not in graph_io_to_skip:
405                            n.type.tensor_type.elem_type = TensorProto.FLOAT16
406                            value_info_list.append(n)
407                    if n.type.HasField("sequence_type"):
408                        if n.type.sequence_type.elem_type.tensor_type.elem_type == TensorProto.FLOAT:
409                            if n.name not in graph_io_to_skip:
410                                n.type.sequence_type.elem_type.tensor_type.elem_type = TensorProto.FLOAT16
411                                value_info_list.append(n)
412
413        queue = next_level
414
415    for value in fp32_initializers.values():
416        # By default, to avoid precision loss, do not convert an initializer to fp16 when it is used only by fp32 nodes.
417        if force_fp16_initializers or value.fp16_nodes:
418            value.initializer = convert_tensor_float_to_float16(value.initializer, min_positive_val, max_finite_val)
419            value_info_list.append(make_value_info_from_tensor(value.initializer))
420            if value.fp32_nodes and not force_fp16_initializers:
421                logger.info(
422                    f"initializer is used by both fp32 and fp16 nodes. Consider add these nodes to block list:{value.fp16_nodes}"
423                )
424
425    # Some operators have data type fixed as float for some input. Add a float16 to float cast for those inputs.
426    for node in mixed_float_type_node_list:
427        for i, input_name in enumerate(node.input):
428            if i not in always_float_inputs[node.op_type] or i in force_fp16_inputs_dict.get(node.op_type, []):
429                continue
430            for value_info in value_info_list:
431                if input_name == value_info.name:
432                    # create new value_info for current node's new input name
433                    new_value_info = model.graph.value_info.add()
434                    new_value_info.CopyFrom(value_info)
435                    output_name = input_name + "_cast_to_fp32"
436                    new_value_info.name = output_name
437                    new_value_info.type.tensor_type.elem_type = TensorProto.FLOAT
438                    # add Cast node (from tensor(float16) to tensor(float) before current node
439                    node_name = input_name + "_cast_to_fp32_node"
440                    new_node = [helper.make_node("Cast", [input_name], [output_name], to=1, name=node_name)]
441                    model.graph.node.extend(new_node)
442                    # change current node's input name
443                    node.input[i] = output_name
444                    break
445
446    accuracy_type = TensorProto.BFLOAT16 if use_bfloat16_as_blocked_nodes_dtype else TensorProto.FLOAT
447    # process the nodes in block list that doesn't support tensor(float16)
448    for node in node_list:
449        # if input's name is in the value_info_list meaning input is tensor(float16) type,
450        # insert a float16 to float Cast node before the node,
451        # change current node's input name and create new value_info for the new name
452        for i in range(len(node.input)):
453            input_name = node.input[i]
454            for value_info in value_info_list:
455                if input_name == value_info.name:
456                    # create new value_info for current node's new input name
457                    new_value_info = model.graph.value_info.add()
458                    new_value_info.CopyFrom(value_info)
459                    output_name = input_name + "_cast_to_fp32"
460                    new_value_info.name = output_name
461                    new_value_info.type.tensor_type.elem_type = accuracy_type
462                    # add Cast node (from tensor(float16) to tensor(float) before current node
463                    node_name = input_name + "_cast_to_fp32_node"
464                    new_node = [helper.make_node("Cast", [input_name], [output_name], to=accuracy_type, name=node_name)]
465                    model.graph.node.extend(new_node)
466                    # change current node's input name
467                    node.input[i] = output_name
468                    break
469        # if output's name is in the value_info_list meaning output is tensor(float16) type, insert a float to
470        # float16 Cast node after the node, change current node's output name and create new value_info for the new name
471        for i in range(len(node.output)):
472            output = node.output[i]
473            for value_info in value_info_list:
474                if output == value_info.name:
475                    # create new value_info for current node's new output
476                    new_value_info = model.graph.value_info.add()
477                    new_value_info.CopyFrom(value_info)
478                    output_cast_name = output + "_cast_to_fp16"
479                    new_value_info.name = output_cast_name
480                    new_value_info.type.tensor_type.elem_type = accuracy_type
481                    # add Cast node (from tensor(float) to tensor(float16) after current node
482                    node_name = output + "_cast_to_fp16_node"
483                    new_node = [helper.make_node("Cast", [output_cast_name], [output], to=10, name=node_name)]
484                    model.graph.node.extend(new_node)
485                    # change current node's output name
486                    node.output[i] = output_cast_name
487                    break
488    return model
489
490
491def float_to_float16_max_diff(tensor, min_positive_val=5.96e-08, max_finite_val=65504.0):
492    """Measure the maximum absolute difference after converting a float tensor to float16."""
493    if not isinstance(tensor, TensorProto):
494        raise ValueError(f"Expected input type is an ONNX TensorProto but got {type(tensor)}")
495    if tensor.data_type != TensorProto.FLOAT:
496        raise ValueError("Expected tensor data type is float.")
497
498    float32_data = None
499    if tensor.float_data:
500        float32_data = np.array(tensor.float_data)
501
502    if tensor.raw_data:
503        float32_data = np.frombuffer(tensor.raw_data, dtype="float32")
504
505    if float32_data is None:
506        raise RuntimeError("external data not loaded!")
507
508    float16_data = convert_np_to_float16(float32_data, min_positive_val, max_finite_val)
509    return np.amax(np.abs(float32_data - np.float32(float16_data)))
510 
codekingpro/portable-devtools · Team Ai