Support int8 quantization in decoder (#1152)

2025-08-08 17:42:21 +00:00 · 2023-06-29 16:48:59 +08:00 · 2023-06-29 16:48:59 +08:00 · db71b03026
commit db71b03026
parent 9c2172c1c4
2 changed files with 2 additions and 2 deletions
--- a/egs/librispeech/ASR/zipformer/export-onnx-streaming.py
+++ b/egs/librispeech/ASR/zipformer/export-onnx-streaming.py
@ -757,7 +757,7 @@ def main():
    quantize_dynamic(
        model_input=decoder_filename,
        model_output=decoder_filename_int8,
-        op_types_to_quantize=["MatMul"],
+        op_types_to_quantize=["MatMul", "Gather"],
        weight_type=QuantType.QInt8,
    )
--- a/egs/librispeech/ASR/zipformer/export-onnx.py
+++ b/egs/librispeech/ASR/zipformer/export-onnx.py
@ -602,7 +602,7 @@ def main():
    quantize_dynamic(
        model_input=decoder_filename,
        model_output=decoder_filename_int8,
-        op_types_to_quantize=["MatMul"],
+        op_types_to_quantize=["MatMul", "Gather"],
        weight_type=QuantType.QInt8,
    )