codekingpro/portable-devtools
114k
1# -------------------------------------------------------------------------
2# Copyright (c) Microsoft Corporation. All rights reserved.
3# Licensed under the MIT License.
4# --------------------------------------------------------------------------
5
6# This file is modified from https://github.com/microsoft/onnxconverter-common/blob/master/onnxconverter_common/float16.py
7# Modifications:
8# (1) Update default value of min_positive_val and max_finite_val
9# (2) keep_io_types can be list of names
10# (3) convert initializers if needed to preserve precision
11# (4) add force_fp16_initializers option
12# (5) handle Resize and GroupNorm with mixed float inputs
13# (6) allow convert_float_to_float16 to accept model path
14
15import itertools
16import logging
17import os
18import tempfile
19
20import numpy as np
21import onnx
22from onnx import AttributeProto, GraphProto, ModelProto, NodeProto, TensorProto, helper, numpy_helper
23from onnx.shape_inference import infer_shapes, infer_shapes_path
24from packaging import version
25
26logger = logging.getLogger(__name__)
27
28
29def _npfloat16_to_int(np_list):
30 """
31 Convert numpy float16 to python int.
32
33 :param np_list: numpy float16 list
34 :return int_list: python int list
35 """
36 return [int(bin(_.view("H"))[2:].zfill(16), 2) for _ in np_list]
37
38
39def convert_np_to_float16(np_array, min_positive_val=5.96e-08, max_finite_val=65504.0):
40 """
41 Convert float32 numpy array to float16 without changing sign or finiteness.
42 Positive values less than min_positive_val are mapped to min_positive_val.
43 Positive finite values greater than max_finite_val are mapped to max_finite_val.
44 Similar for negative values. NaN, 0, inf, and -inf are unchanged.
45 """
46
47 def between(a, b, c):
48 return np.logical_and(a < b, b < c)
49
50 if np_array[np.where(np_array > 0)].shape[0] > 0:
51 positive_max = np_array[np.where(np_array > 0)].max()
52 positive_min = np_array[np.where(np_array > 0)].min()
53 if positive_max >= max_finite_val:
54 logger.debug(f"the float32 number {positive_max} will be truncated to {max_finite_val}")
55 if positive_min <= min_positive_val:
56 logger.debug(f"the float32 number {positive_min} will be truncated to {min_positive_val}")
57
58 if np_array[np.where(np_array < 0)].shape[0] > 0:
59 negative_max = np_array[np.where(np_array < 0)].max()
60 negative_min = np_array[np.where(np_array < 0)].min()
61 if negative_min <= -max_finite_val:
62 logger.debug(f"the float32 number {negative_min} will be truncated to {-max_finite_val}")
63 if negative_max >= -min_positive_val:
64 logger.debug(f"the float32 number {negative_max} will be truncated to {-min_positive_val}")
65
66 np_array = np.where(between(0, np_array, min_positive_val), min_positive_val, np_array)
67 np_array = np.where(between(-min_positive_val, np_array, 0), -min_positive_val, np_array)
68 np_array = np.where(between(max_finite_val, np_array, float("inf")), max_finite_val, np_array)
69 np_array = np.where(between(float("-inf"), np_array, -max_finite_val), -max_finite_val, np_array)
70 return np.float16(np_array)
71
72
73def convert_tensor_float_to_float16(tensor, min_positive_val=5.96e-08, max_finite_val=65504.0):
74 """Convert tensor float to float16.
75
76 Args:
77 tensor (TensorProto): the tensor to convert.
78 min_positive_val (float, optional): minimal positive value. Defaults to 1e-7.
79 max_finite_val (float, optional): maximal finite value. Defaults to 1e4.
80
81 Raises:
82 ValueError: input type is not TensorProto.
83
84 Returns:
85 TensorProto: the converted tensor.
86 """
87
88 if not isinstance(tensor, TensorProto):
89 raise ValueError(f"Expected input type is an ONNX TensorProto but got {type(tensor)}")
90
91 if tensor.data_type == TensorProto.FLOAT:
92 tensor.data_type = TensorProto.FLOAT16
93 # convert float_data (float type) to float16 and write to int32_data
94 if tensor.float_data:
95 float16_data = convert_np_to_float16(np.array(tensor.float_data), min_positive_val, max_finite_val)
96 int_list = _npfloat16_to_int(float16_data)
97 tensor.int32_data[:] = int_list
98 tensor.float_data[:] = []
99 # convert raw_data (bytes type)
100 if tensor.raw_data:
101 # convert n.raw_data to float
102 float32_list = np.frombuffer(tensor.raw_data, dtype="float32")
103 # convert float to float16
104 float16_list = convert_np_to_float16(float32_list, min_positive_val, max_finite_val)
105 # convert float16 to bytes and write back to raw_data
106 tensor.raw_data = float16_list.tobytes()
107 return tensor
108
109
110def make_value_info_from_tensor(tensor):
111 shape = numpy_helper.to_array(tensor).shape
112 return helper.make_tensor_value_info(tensor.name, tensor.data_type, shape)
113
114
115DEFAULT_OP_BLOCK_LIST = [
116 "ArrayFeatureExtractor",
117 "Binarizer",
118 "CastMap",
119 "CategoryMapper",
120 "DictVectorizer",
121 "FeatureVectorizer",
122 "Imputer",
123 "LabelEncoder",
124 "LinearClassifier",
125 "LinearRegressor",
126 "Normalizer",
127 "OneHotEncoder",
128 "RandomUniformLike",
129 "SVMClassifier",
130 "SVMRegressor",
131 "Scaler",
132 "TreeEnsembleClassifier",
133 "TreeEnsembleRegressor",
134 "TreeEnsemble",
135 "ZipMap",
136 "NonMaxSuppression",
137 "TopK",
138 "RoiAlign",
139 "Range",
140 "CumSum",
141 "Min",
142 "Max",
143 "Upsample",
144]
145
146
147# Some operators has data type fixed as float for some inputs. Key is op_type, value is list of input indices
148# Note that DirectML allows float16 gamma and beta in GroupNorm. Use force_fp16_inputs parameter could overwrite this.
149ALWAYS_FLOAT_INPUTS = {"Resize": [2], "GroupNorm": [1, 2], "SkipGroupNorm": [1, 2]}
150
151
152class InitializerTracker:
153 """Class for keeping track of initializer."""
154
155 def __init__(self, initializer: TensorProto):
156 self.initializer = initializer
157 self.fp32_nodes = []
158 self.fp16_nodes = []
159
160 def add_node(self, node: NodeProto, is_node_blocked):
161 if is_node_blocked:
162 self.fp32_nodes.append(node)
163 else:
164 self.fp16_nodes.append(node)
165
166
167def convert_float_to_float16(
168 model,
169 min_positive_val=5.96e-08,
170 max_finite_val=65504.0,
171 keep_io_types=False,
172 disable_shape_infer=False,
173 op_block_list=None,
174 node_block_list=None,
175 force_fp16_initializers=False,
176 force_fp16_inputs=None,
177 use_bfloat16_as_blocked_nodes_dtype=False,
178):
179 """Convert tensor float type in the input ONNX model to tensor float16.
180
181 Args:
182 model (ModelProto or str): The ONNX model or path of the model to convert.
183 min_positive_val (float, optional): minimal positive value. Defaults to 5.96e-08.
184 max_finite_val (float, optional): maximal finite value of float16. Defaults to 65504.
185 keep_io_types (Union[bool, List[str]], optional): It could be boolean or a list of float32 input/output names.
186 If True, model inputs/outputs should be left as float32.
187 Defaults to False.
188 disable_shape_infer (bool, optional): Skips running onnx shape/type inference.
189 Useful if shape inference has been done. Defaults to False.
190 op_block_list (List[str], optional): List of op types to leave as float32.
191 Defaults to None, which will use `float16.DEFAULT_OP_BLOCK_LIST`.
192 node_block_list (List[str], optional): List of node names to leave as float32. Defaults to None.
193 force_fp16_initializers(bool): force converting all float initializers to float16.
194 Default to false, which will convert only the one needed to avoid precision loss.
195 force_fp16_inputs(Dict[str, List[int]]): Force the conversion of the inputs of some operators to float16, even if
196 this script's preference it to keep them in float32.
197 Raises:
198 ValueError: input type is not ModelProto.
199
200 Returns:
201 ModelProto: converted model.
202 """
203 assert min_positive_val >= 5.96e-08, (
204 "invalid min_positive_val. smallest positive float16 value: subnormal 5.96e-08, and normalized 6.104e-05"
205 )
206 assert max_finite_val <= float(np.finfo(np.float16).max), "invalid max_finite_val. largest float16 value: 65504"
207
208 force_fp16_inputs_dict = {} if force_fp16_inputs is None else force_fp16_inputs
209
210 if isinstance(model, str):
211 model_path = model
212 if version.parse(onnx.__version__) >= version.parse("1.8.0") and not disable_shape_infer:
213 # shape_infer_model_path should be in the same folder of model_path
214 with tempfile.NamedTemporaryFile(dir=os.path.dirname(model_path)) as tmpfile:
215 shape_infer_model_path = tmpfile.name
216 # infer_shapes_path can be used for model >2GB, and infer_shapes cannot.
217 infer_shapes_path(model_path, shape_infer_model_path)
218 model = onnx.load(shape_infer_model_path)
219 disable_shape_infer = True
220 else:
221 model = onnx.load(model_path)
222
223 if not isinstance(model, ModelProto):
224 raise ValueError(f"Expected an ONNX ModelProto but got {type(model)}")
225
226 func_infer_shape = None
227 if not disable_shape_infer and version.parse(onnx.__version__) >= version.parse("1.2.0"):
228 try:
229 func_infer_shape = infer_shapes
230 finally:
231 pass
232
233 # create blocklists
234 if op_block_list is None:
235 op_block_list = DEFAULT_OP_BLOCK_LIST
236 if node_block_list is None:
237 node_block_list = []
238 op_block_list = set(op_block_list)
239 node_block_list = set(node_block_list)
240
241 # Build opset-aware always_float_inputs: Resize input layout differs between opset 10 and 11+.
242 # Opset 10: [X, scales] — scales at index 1 must stay float32.
243 # Opset 11+: [X, roi, scales, sizes] — scales at index 2 must stay float32; roi (index 1) allows fp16.
244 onnx_opset = max((o.version for o in model.opset_import if o.domain in ("", "ai.onnx")), default=11)
245 always_float_inputs = dict(ALWAYS_FLOAT_INPUTS)
246 if onnx_opset <= 10:
247 always_float_inputs["Resize"] = [1]
248
249 logger.debug(
250 f"fp16 parameters: min_positive_val={min_positive_val} max_finite_val={max_finite_val} keep_io_types={keep_io_types} disable_shape_infer={disable_shape_infer} op_block_list={op_block_list} node_block_list={node_block_list} force_fp16_initializers={force_fp16_initializers}"
251 )
252
253 # create a queue for BFS
254 queue = []
255 value_info_list = []
256 node_list = []
257
258 # Some operators (Like Resize or GroupNorm) have data type fixed as float for some input.
259 # When it is converted to float16, there are mixed types: some inputs are float32 and some are float16.
260 # This list keeps track of such nodes that are not in block list.
261 mixed_float_type_node_list = []
262
263 # type inference on input model
264 if func_infer_shape is not None:
265 model = func_infer_shape(model)
266 queue.append(model)
267 name_mapping = {}
268 graph_io_to_skip = set()
269 io_casts = set()
270
271 fp32_inputs = [n.name for n in model.graph.input if n.type.tensor_type.elem_type == TensorProto.FLOAT]
272 fp32_outputs = [n.name for n in model.graph.output if n.type.tensor_type.elem_type == TensorProto.FLOAT]
273 if isinstance(keep_io_types, list):
274 fp32_inputs = [n for n in fp32_inputs if n in keep_io_types]
275 fp32_outputs = [n for n in fp32_outputs if n in keep_io_types]
276 elif not keep_io_types:
277 fp32_inputs = []
278 fp32_outputs = []
279
280 for i, n in enumerate(model.graph.input):
281 if n.name in fp32_inputs:
282 output_name = "graph_input_cast_" + str(i)
283 name_mapping[n.name] = output_name
284 graph_io_to_skip.add(n.name)
285
286 node_name = "graph_input_cast" + str(i)
287 new_value_info = model.graph.value_info.add()
288 new_value_info.CopyFrom(n)
289 new_value_info.name = output_name
290 new_value_info.type.tensor_type.elem_type = TensorProto.FLOAT16
291 # add Cast node (from tensor(float) to tensor(float16) after graph input
292 new_node = [helper.make_node("Cast", [n.name], [output_name], to=TensorProto.FLOAT16, name=node_name)]
293 model.graph.node.extend(new_node)
294 value_info_list.append(new_value_info)
295 io_casts.add(node_name)
296
297 for i, n in enumerate(model.graph.output):
298 if n.name in fp32_outputs:
299 input_name = "graph_output_cast_" + str(i)
300 name_mapping[n.name] = input_name
301 graph_io_to_skip.add(n.name)
302
303 node_name = "graph_output_cast" + str(i)
304 # add Cast node (from tensor(float16) to tensor(float) before graph output
305 new_value_info = model.graph.value_info.add()
306 new_value_info.CopyFrom(n)
307 new_value_info.name = input_name
308 new_value_info.type.tensor_type.elem_type = TensorProto.FLOAT16
309 new_node = [helper.make_node("Cast", [input_name], [n.name], to=1, name=node_name)]
310 model.graph.node.extend(new_node)
311 value_info_list.append(new_value_info)
312 io_casts.add(node_name)
313
314 fp32_initializers: dict[str, InitializerTracker] = {}
315 while queue:
316 next_level = []
317 for q in queue:
318 # if q is model, push q.graph (GraphProto)
319 if isinstance(q, ModelProto):
320 next_level.append(q.graph)
321 # if q is model.graph, push q.node.attribute (AttributeProto)
322 if isinstance(q, GraphProto):
323 for n in q.initializer: # TensorProto type
324 if n.data_type == TensorProto.FLOAT:
325 assert n.name not in fp32_initializers
326 fp32_initializers[n.name] = InitializerTracker(n)
327
328 for n in q.node:
329 # if n is in the block list (doesn't support float16), no conversion for the node,
330 # and save the node for further processing
331 if n.name in io_casts:
332 continue
333 for i in range(len(n.input)):
334 if n.input[i] in name_mapping:
335 n.input[i] = name_mapping[n.input[i]]
336 for i in range(len(n.output)):
337 if n.output[i] in name_mapping:
338 n.output[i] = name_mapping[n.output[i]]
339
340 is_node_blocked = n.op_type in op_block_list or n.name in node_block_list
341 for i, input_name in enumerate(n.input):
342 if input_name in fp32_initializers:
343 # For Resize/GroupNorm, only the first input can be float16
344 use_fp32_weight = is_node_blocked or (
345 i in always_float_inputs.get(n.op_type, [])
346 and i not in force_fp16_inputs_dict.get(n.op_type, [])
347 )
348 fp32_initializers[input_name].add_node(n, use_fp32_weight)
349
350 if is_node_blocked:
351 node_list.append(n)
352 else:
353 if n.op_type == "Cast":
354 for attr in n.attribute:
355 if attr.name == "to" and attr.i == TensorProto.FLOAT:
356 attr.i = TensorProto.FLOAT16
357 break
358
359 if n.op_type in [
360 "EyeLike",
361 "Multinomial",
362 "RandomNormal",
363 "RandomNormalLike",
364 "RandomUniform",
365 "RandomUniformLike",
366 "SequenceEmpty",
367 "Bernoulli",
368 ]:
369 has_dtype = False
370 for attr in n.attribute:
371 if attr.name == "dtype":
372 has_dtype = True
373 if attr.i == TensorProto.FLOAT:
374 attr.i = TensorProto.FLOAT16
375
376 # The dtype attribute is optional and default is FLOAT in the following operators
377 # so we need add dtype attribute to specify the data type float16
378 if (n.op_type in ["RandomNormal", "RandomUniform", "SequenceEmpty"]) and not has_dtype:
379 n.attribute.extend([helper.make_attribute("dtype", TensorProto.FLOAT16)])
380
381 # For Resize/GroupNorm, attribute data type cannot be changed
382 if n.op_type not in always_float_inputs or n.op_type in force_fp16_inputs_dict:
383 for attr in n.attribute:
384 next_level.append(attr) # noqa: PERF402
385 else:
386 mixed_float_type_node_list.append(n)
387
388 # if q is model.graph.node.attribute, push q.g and q.graphs (GraphProto)
389 # and process node.attribute.t and node.attribute.tensors (TensorProto)
390 if isinstance(q, AttributeProto):
391 next_level.append(q.g)
392 for n in q.graphs:
393 next_level.append(n) # noqa: PERF402
394 q.t.CopyFrom(convert_tensor_float_to_float16(q.t, min_positive_val, max_finite_val))
395 for n in q.tensors:
396 n = convert_tensor_float_to_float16(n, min_positive_val, max_finite_val) # noqa: PLW2901
397 # if q is graph, process input, output and value_info (ValueInfoProto)
398 if isinstance(q, GraphProto):
399 # Note that float initializers tracked by fp32_initializers will be processed later.
400 # for all ValueInfoProto with tensor(float) type in input, output and value_info, convert them to
401 # tensor(float16) except map and seq(map). And save them in value_info_list for further processing
402 for n in itertools.chain(q.input, q.output, q.value_info):
403 if n.type.tensor_type.elem_type == TensorProto.FLOAT:
404 if n.name not in graph_io_to_skip:
405 n.type.tensor_type.elem_type = TensorProto.FLOAT16
406 value_info_list.append(n)
407 if n.type.HasField("sequence_type"):
408 if n.type.sequence_type.elem_type.tensor_type.elem_type == TensorProto.FLOAT:
409 if n.name not in graph_io_to_skip:
410 n.type.sequence_type.elem_type.tensor_type.elem_type = TensorProto.FLOAT16
411 value_info_list.append(n)
412
413 queue = next_level
414
415 for value in fp32_initializers.values():
416 # By default, to avoid precision loss, do not convert an initializer to fp16 when it is used only by fp32 nodes.
417 if force_fp16_initializers or value.fp16_nodes:
418 value.initializer = convert_tensor_float_to_float16(value.initializer, min_positive_val, max_finite_val)
419 value_info_list.append(make_value_info_from_tensor(value.initializer))
420 if value.fp32_nodes and not force_fp16_initializers:
421 logger.info(
422 f"initializer is used by both fp32 and fp16 nodes. Consider add these nodes to block list:{value.fp16_nodes}"
423 )
424
425 # Some operators have data type fixed as float for some input. Add a float16 to float cast for those inputs.
426 for node in mixed_float_type_node_list:
427 for i, input_name in enumerate(node.input):
428 if i not in always_float_inputs[node.op_type] or i in force_fp16_inputs_dict.get(node.op_type, []):
429 continue
430 for value_info in value_info_list:
431 if input_name == value_info.name:
432 # create new value_info for current node's new input name
433 new_value_info = model.graph.value_info.add()
434 new_value_info.CopyFrom(value_info)
435 output_name = input_name + "_cast_to_fp32"
436 new_value_info.name = output_name
437 new_value_info.type.tensor_type.elem_type = TensorProto.FLOAT
438 # add Cast node (from tensor(float16) to tensor(float) before current node
439 node_name = input_name + "_cast_to_fp32_node"
440 new_node = [helper.make_node("Cast", [input_name], [output_name], to=1, name=node_name)]
441 model.graph.node.extend(new_node)
442 # change current node's input name
443 node.input[i] = output_name
444 break
445
446 accuracy_type = TensorProto.BFLOAT16 if use_bfloat16_as_blocked_nodes_dtype else TensorProto.FLOAT
447 # process the nodes in block list that doesn't support tensor(float16)
448 for node in node_list:
449 # if input's name is in the value_info_list meaning input is tensor(float16) type,
450 # insert a float16 to float Cast node before the node,
451 # change current node's input name and create new value_info for the new name
452 for i in range(len(node.input)):
453 input_name = node.input[i]
454 for value_info in value_info_list:
455 if input_name == value_info.name:
456 # create new value_info for current node's new input name
457 new_value_info = model.graph.value_info.add()
458 new_value_info.CopyFrom(value_info)
459 output_name = input_name + "_cast_to_fp32"
460 new_value_info.name = output_name
461 new_value_info.type.tensor_type.elem_type = accuracy_type
462 # add Cast node (from tensor(float16) to tensor(float) before current node
463 node_name = input_name + "_cast_to_fp32_node"
464 new_node = [helper.make_node("Cast", [input_name], [output_name], to=accuracy_type, name=node_name)]
465 model.graph.node.extend(new_node)
466 # change current node's input name
467 node.input[i] = output_name
468 break
469 # if output's name is in the value_info_list meaning output is tensor(float16) type, insert a float to
470 # float16 Cast node after the node, change current node's output name and create new value_info for the new name
471 for i in range(len(node.output)):
472 output = node.output[i]
473 for value_info in value_info_list:
474 if output == value_info.name:
475 # create new value_info for current node's new output
476 new_value_info = model.graph.value_info.add()
477 new_value_info.CopyFrom(value_info)
478 output_cast_name = output + "_cast_to_fp16"
479 new_value_info.name = output_cast_name
480 new_value_info.type.tensor_type.elem_type = accuracy_type
481 # add Cast node (from tensor(float) to tensor(float16) after current node
482 node_name = output + "_cast_to_fp16_node"
483 new_node = [helper.make_node("Cast", [output_cast_name], [output], to=10, name=node_name)]
484 model.graph.node.extend(new_node)
485 # change current node's output name
486 node.output[i] = output_cast_name
487 break
488 return model
489
490
491def float_to_float16_max_diff(tensor, min_positive_val=5.96e-08, max_finite_val=65504.0):
492 """Measure the maximum absolute difference after converting a float tensor to float16."""
493 if not isinstance(tensor, TensorProto):
494 raise ValueError(f"Expected input type is an ONNX TensorProto but got {type(tensor)}")
495 if tensor.data_type != TensorProto.FLOAT:
496 raise ValueError("Expected tensor data type is float.")
497
498 float32_data = None
499 if tensor.float_data:
500 float32_data = np.array(tensor.float_data)
501
502 if tensor.raw_data:
503 float32_data = np.frombuffer(tensor.raw_data, dtype="float32")
504
505 if float32_data is None:
506 raise RuntimeError("external data not loaded!")
507
508 float16_data = convert_np_to_float16(float32_data, min_positive_val, max_finite_val)
509 return np.amax(np.abs(float32_data - np.float32(float16_data)))
510 