codekingpro/portable-devtools
115k
1import onnx
2import onnx.helper
3
4from ..quant_utils import TENSOR_NAME_QUANT_SUFFIX, QuantizedValue, QuantizedValueType, attribute_to_kwarg, ms_domain
5from .base_operator import QuantOperatorBase
6
7
8class QLinearSoftmax(QuantOperatorBase):
9 def quantize(self):
10 node = self.node
11 # set limitations for softmax output scale and zp, because the output of softmax is always 0-1
12 if self.quantizer.activation_qType == onnx.onnx_pb.TensorProto.UINT8:
13 out_scale = 1 / 256.0
14 out_zero_point = 0
15 else:
16 out_scale = 1 / 256.0
17 out_zero_point = -128
18 # only try to quantize when given quantization parameters for it
19 (
20 data_found,
21 output_scale_name,
22 output_zp_name,
23 _,
24 _,
25 ) = self.quantizer._get_quantization_params(node.output[0], out_scale, out_zero_point)
26
27 # get quantized input tensor names, quantize input if needed
28 (
29 quantized_input_names,
30 input_zero_point_names,
31 input_scale_names,
32 nodes,
33 ) = self.quantizer.quantize_activation(node, [0])
34
35 if not data_found or quantized_input_names is None:
36 return super().quantize()
37
38 # Create an entry for output quantized value.
39 qlinear_output_name = node.output[0] + TENSOR_NAME_QUANT_SUFFIX
40 quantized_output_value = QuantizedValue(
41 node.output[0],
42 qlinear_output_name,
43 output_scale_name,
44 output_zp_name,
45 QuantizedValueType.Input,
46 )
47 self.quantizer.quantized_value_map[node.output[0]] = quantized_output_value
48
49 # Create qlinear softmax node for given type
50 kwargs = {}
51 for attribute in node.attribute:
52 kwargs.update(attribute_to_kwarg(attribute))
53 kwargs["domain"] = ms_domain
54 # make qlinearsoft has the real opset_version, its default SinceVersion would be 1
55 kwargs["opset"] = self.quantizer.opset_version
56 qlinear_node_name = node.name + "_quant" if node.name else ""
57 qnode = onnx.helper.make_node(
58 "QLinear" + node.op_type,
59 [
60 quantized_input_names[0],
61 input_scale_names[0],
62 input_zero_point_names[0],
63 output_scale_name,
64 output_zp_name,
65 ],
66 [qlinear_output_name],
67 qlinear_node_name,
68 **kwargs,
69 )
70
71 # add all newly created nodes
72 nodes.append(qnode)
73 self.quantizer.new_nodes += nodes
74 return None
75 