SDNQ enable matmul support for float8_e5m2

This commit is contained in:
Disty0
2025-06-02 00:53:10 +03:00
parent 8f1a1d7311
commit e8588c91ea
2 changed files with 3 additions and 3 deletions
+2 -2
View File
@@ -28,9 +28,9 @@ dtype_dict = {
"float8_e5m2fnuz": {"min": -57344, "max": 57344, "num_bits": 8, "target_dtype": CustomDtype.FP8, "torch_dtype": torch.float8_e5m2fnuz, "storage_dtype": torch.float8_e5m2fnuz, "is_unsigned": False, "is_integer": False},
}
quantized_matmul_dtypes = ("int8", "int6", "int4", "int2", "float8_e4m3fn")
quantized_matmul_dtypes = ("int8", "int6", "int4", "int2", "float8_e4m3fn", "float8_e5m2")
if devices.backend in {"cpu", "openvino"}:
quantized_matmul_dtypes += ("float8_e5m2", "float8_e4m3fnuz", "float8_e5m2fnuz")
quantized_matmul_dtypes += ("float8_e4m3fnuz", "float8_e5m2fnuz")
linear_types = ("Linear",)
conv_types = ("Conv1d", "Conv2d", "Conv3d")