codekingpro/portable-devtools
114k
1# --------------------------------------------------------------------------
2# Copyright (c) Microsoft Corporation. All rights reserved.
3# Licensed under the MIT License.
4# --------------------------------------------------------------------------
5from __future__ import annotations
6
7from typing import Any
8
9import numpy as np
10import onnx
11
12from ..quant_utils import (
13 TENSOR_NAME_QUANT_SUFFIX,
14 QuantizedValue,
15 QuantizedValueType,
16 attribute_to_kwarg,
17 quantize_nparray,
18)
19from .base_operator import QuantOperatorBase
20from .qdq_base_operator import QDQOperatorBase
21
22
23class QPad(QuantOperatorBase):
24 def __init__(self, onnx_quantizer, onnx_node):
25 super().__init__(onnx_quantizer, onnx_node)
26
27 def quantize(self):
28 node = self.node
29 assert node.op_type == "Pad"
30
31 # Only after version 11, it has the optional constant_value
32 # If input[0] is not quantized, do not quanitize this node
33 if (self.quantizer.opset_version < 11) or (node.input[0] not in self.quantizer.quantized_value_map):
34 super().quantize()
35 return
36 quantized_input_value = self.quantizer.quantized_value_map[node.input[0]]
37
38 kwargs = {}
39 for attribute in node.attribute:
40 kv = attribute_to_kwarg(attribute)
41 kwargs.update(kv)
42
43 if "mode" not in kwargs or kwargs["mode"] == b"constant":
44 if len(node.input) > 2 and node.input[2] != "": # There is 3rd input 'constant_value'
45 zp_tensor = self.quantizer.model.get_initializer(quantized_input_value.zp_name)
46 scale_tensor = self.quantizer.model.get_initializer(quantized_input_value.scale_name)
47 if zp_tensor is None or scale_tensor is None:
48 super().quantize()
49 return
50
51 padding_constant_initializer = self.quantizer.model.get_initializer(node.input[2])
52 if padding_constant_initializer is not None:
53 zp_array = onnx.numpy_helper.to_array(zp_tensor)
54 zp_value = zp_array.item() if zp_array.ndim == 0 else zp_array[0]
55 scale_array = onnx.numpy_helper.to_array(scale_tensor)
56 scale_value = scale_array.item() if scale_array.ndim == 0 else scale_array[0]
57 padding_constant_array = onnx.numpy_helper.to_array(padding_constant_initializer)
58 quantized_padding_constant_array = quantize_nparray(
59 self.quantizer.activation_qType,
60 padding_constant_array,
61 scale_value,
62 zp_value,
63 )
64 quantized_padding_constant_name = node.input[2] + TENSOR_NAME_QUANT_SUFFIX
65 quantized_padding_constant_initializer = onnx.numpy_helper.from_array(
66 quantized_padding_constant_array,
67 quantized_padding_constant_name,
68 )
69 # Suppose this padding constant initializer only used by the node
70 self.quantizer.model.remove_initializer(padding_constant_initializer)
71 self.quantizer.model.add_initializer(quantized_padding_constant_initializer)
72 node.input[2] = quantized_padding_constant_name
73 else:
74 # TODO: check quantize_inputs after sub graph is supported
75 pad_value_qnodes = self.quantizer._get_quantize_input_nodes(
76 node,
77 2,
78 self.quantizer.activation_qType,
79 quantized_input_value.scale_name,
80 quantized_input_value.zp_name,
81 initial_type=scale_tensor.data_type,
82 )
83 self.quantizer.new_nodes.extend(pad_value_qnodes)
84 node.input[2] = pad_value_qnodes[0].output[0]
85 else:
86 # In quantized format, the `zero` before quantization is mapped
87 # to quantized_input_value.zp_name. Thus, padding 0 to
88 # original tensor should become padding zero point to quantized
89 # tensor.
90 if len(node.input) == 2:
91 # Feed quantization's zero point to padding node.
92 node.input.append(quantized_input_value.zp_name)
93 else:
94 # Assign quantization's zero point to padding node.
95 assert node.input[2] == ""
96 node.input[2] = quantized_input_value.zp_name
97
98 # Create an entry for output quantized value
99 quantized_output_value = QuantizedValue(
100 node.output[0],
101 node.output[0] + TENSOR_NAME_QUANT_SUFFIX,
102 quantized_input_value.scale_name,
103 quantized_input_value.zp_name,
104 QuantizedValueType.Input,
105 )
106 self.quantizer.quantized_value_map[node.output[0]] = quantized_output_value
107
108 node.input[0] = quantized_input_value.q_name
109 node.output[0] = quantized_output_value.q_name
110 self.quantizer.new_nodes += [node]
111
112
113class QDQPad(QDQOperatorBase):
114 def __init__(self, onnx_quantizer, onnx_node):
115 super().__init__(onnx_quantizer, onnx_node)
116
117 def _get_pad_const_val(self, attrs_dict: dict[str, Any]) -> np.ndarray | None:
118 """
119 Returns the Pad's constant padding value. Returns `None` if the padding value is
120 not constant (i.e., comes from a dynamic input).
121 """
122 const_val = None
123 onnx_tensor_type = self.quantizer.model.get_tensor_type(self.node.input[0])
124 if onnx_tensor_type is None:
125 return None
126
127 np_dtype = onnx.helper.tensor_dtype_to_np_dtype(onnx_tensor_type.elem_type)
128 if self.quantizer.opset_version < 11:
129 const_val = np.array(attrs_dict.get("value", 0), dtype=np_dtype)
130 elif len(self.node.input) >= 3 and self.node.input[2]:
131 const_val = self.quantizer.model.get_constant_value(self.node.input[2])
132 else:
133 const_val = np.array(0, dtype=np_dtype)
134
135 return const_val
136
137 def _should_quantize_output_same_as_input(self) -> bool:
138 """
139 Returns true if Pad's output should use the same quantization parameters as input[0]
140 """
141 attrs_dict = {}
142 for attribute in self.node.attribute:
143 kv = attribute_to_kwarg(attribute)
144 attrs_dict.update(kv)
145
146 pad_mode = attrs_dict.get("mode", b"constant")
147 if pad_mode in (b"reflect", b"edge", b"wrap"):
148 # These modes pad the output with a value that already exists in the input.
149 # So, we can quantize the output the same as the input.
150 return True
151
152 # For 'constant' mode, if padding with 0, we can also quantize the output the same as the input
153 # because our quantization floating-point range always includes 0.
154 if pad_mode == b"constant":
155 pad_val = self._get_pad_const_val(attrs_dict)
156 if pad_val is not None and pad_val.dtype in (np.float32, np.float16):
157 return float(pad_val.item()) == 0
158
159 return False
160
161 def quantize(self):
162 assert self.node.op_type == "Pad"
163
164 for input_name in self.node.input:
165 if input_name:
166 self.quantizer.quantize_activation_tensor(input_name)
167
168 if not self.disable_qdq_for_node_output:
169 if self._should_quantize_output_same_as_input():
170 self.quantizer.quantize_output_same_as_input(self.node.output[0], self.node.input[0], self.node.name)
171 else:
172 self.quantizer.quantize_activation_tensor(self.node.output[0])
173 