diff --git a/modules/intel/ipex/hijacks.py b/modules/intel/ipex/hijacks.py index d81d7b05c..2a3b9e06a 100644 --- a/modules/intel/ipex/hijacks.py +++ b/modules/intel/ipex/hijacks.py @@ -254,8 +254,15 @@ torch.Tensor.original_Tensor_to = torch.Tensor.to @wraps(torch.Tensor.to) def Tensor_to(self, device=None, *args, **kwargs): if check_cuda(device): + if not device_supports_fp64 and kwargs.get("dtype", None) == torch.float64: + kwargs["dtype"] = torch.float32 return self.original_Tensor_to(return_xpu(device), *args, **kwargs) else: + if not device_supports_fp64: + if kwargs.get("dtype", None) == torch.float64 and torch.device(device).type == "xpu": + kwargs["dtype"] = torch.float32 + elif device == torch.float64 and self.device.type == "xpu": + device = torch.float32 return self.original_Tensor_to(device, *args, **kwargs) original_Tensor_cuda = torch.Tensor.cuda diff --git a/modules/sdnq/__init__.py b/modules/sdnq/__init__.py index d9db525db..998e0a467 100644 --- a/modules/sdnq/__init__.py +++ b/modules/sdnq/__init__.py @@ -56,8 +56,11 @@ def sdnq_quantize_layer(layer, weights_dtype="int8", torch_dtype=None, group_siz output_channel_size, channel_size = layer.weight.shape if use_quantized_matmul: use_quantized_matmul = weights_dtype in quantized_matmul_dtypes and channel_size >= 32 and output_channel_size >= 32 - if use_quantized_matmul and not dtype_dict[weights_dtype]["is_integer"]: - use_quantized_matmul = output_channel_size % 16 == 0 and channel_size % 16 == 0 + if use_quantized_matmul: + if dtype_dict[weights_dtype]["is_integer"]: + use_quantized_matmul = output_channel_size % 8 == 0 and channel_size % 8 == 0 + else: + use_quantized_matmul = output_channel_size % 16 == 0 and channel_size % 16 == 0 if group_size == 0: if is_linear_type: