From e657cf790d2dd876510ff48b4c68c66425965af6 Mon Sep 17 00:00:00 2001 From: Disty0 Date: Wed, 18 Jun 2025 02:12:34 +0300 Subject: [PATCH] SDNQ fix int8 matmul with qwen --- modules/sdnq/__init__.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/modules/sdnq/__init__.py b/modules/sdnq/__init__.py index d9db525db..998e0a467 100644 --- a/modules/sdnq/__init__.py +++ b/modules/sdnq/__init__.py @@ -56,8 +56,11 @@ def sdnq_quantize_layer(layer, weights_dtype="int8", torch_dtype=None, group_siz output_channel_size, channel_size = layer.weight.shape if use_quantized_matmul: use_quantized_matmul = weights_dtype in quantized_matmul_dtypes and channel_size >= 32 and output_channel_size >= 32 - if use_quantized_matmul and not dtype_dict[weights_dtype]["is_integer"]: - use_quantized_matmul = output_channel_size % 16 == 0 and channel_size % 16 == 0 + if use_quantized_matmul: + if dtype_dict[weights_dtype]["is_integer"]: + use_quantized_matmul = output_channel_size % 8 == 0 and channel_size % 8 == 0 + else: + use_quantized_matmul = output_channel_size % 16 == 0 and channel_size % 16 == 0 if group_size == 0: if is_linear_type: