Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
direct_q8.py79 linesDownload Raw Back to operators
1from ..quant_utils import TENSOR_NAME_QUANT_SUFFIX, QuantizedValue, QuantizedValueType
2from .base_operator import QuantOperatorBase
3from .qdq_base_operator import QDQOperatorBase
4
5
6# For operators that support 8bits operations directly, and output could
7# reuse input[0]'s type, zeropoint, scale; For example,Transpose, Reshape, etc.
8class Direct8BitOp(QuantOperatorBase):
9    def __init__(self, onnx_quantizer, onnx_node):
10        super().__init__(onnx_quantizer, onnx_node)
11
12    def quantize(self):
13        node = self.node
14
15        if not self.quantizer.force_quantize_no_input_check:
16            # Keep backward compatibility
17            # Quantize when input[0] is quantized already. Otherwise keep it.
18            quantized_input_value = self.quantizer.find_quantized_value(node.input[0])
19            if quantized_input_value is None:
20                self.quantizer.new_nodes += [node]
21                return
22
23            quantized_output_value = QuantizedValue(
24                node.output[0],
25                node.output[0] + TENSOR_NAME_QUANT_SUFFIX,
26                quantized_input_value.scale_name,
27                quantized_input_value.zp_name,
28                quantized_input_value.value_type,
29            )
30            self.quantizer.quantized_value_map[node.output[0]] = quantized_output_value
31
32            node.input[0] = quantized_input_value.q_name
33            node.output[0] = quantized_output_value.q_name
34            self.quantizer.new_nodes += [node]
35
36        else:
37            # Force quantize those ops if possible, use exclude node list if this is not you want
38            if not self.quantizer.is_valid_quantize_weight(node.input[0]):
39                super().quantize()
40                return
41
42            (
43                quantized_input_names,
44                zero_point_names,
45                scale_names,
46                nodes,
47            ) = self.quantizer.quantize_activation(node, [0])
48            if quantized_input_names is None:
49                return super().quantize()
50
51            # Create an entry for output quantized value
52            quantized_output_value = QuantizedValue(
53                node.output[0],
54                node.output[0] + TENSOR_NAME_QUANT_SUFFIX,
55                scale_names[0],
56                zero_point_names[0],
57                QuantizedValueType.Input,
58            )
59            self.quantizer.quantized_value_map[node.output[0]] = quantized_output_value
60
61            node.input[0] = quantized_input_names[0]
62            node.output[0] = quantized_output_value.q_name
63            nodes.append(node)
64
65            self.quantizer.new_nodes += nodes
66
67
68class QDQDirect8BitOp(QDQOperatorBase):
69    def __init__(self, onnx_quantizer, onnx_node):
70        super().__init__(onnx_quantizer, onnx_node)
71
72    def quantize(self):
73        if self.quantizer.force_quantize_no_input_check:
74            self.quantizer.quantize_activation_tensor(self.node.input[0])
75            if not self.disable_qdq_for_node_output:
76                self.quantizer.quantize_output_same_as_input(self.node.output[0], self.node.input[0], self.node.name)
77        elif self.quantizer.is_tensor_quantized(self.node.input[0]) and not self.disable_qdq_for_node_output:
78            self.quantizer.quantize_output_same_as_input(self.node.output[0], self.node.input[0], self.node.name)
79 
codekingpro/portable-devtools · Team Ai