Update changelog

This commit is contained in:
Disty0
2025-05-29 03:26:47 +03:00
parent 67e0f4d833
commit 2cc5a58b0f
2 changed files with 3 additions and 1 deletions
+2
View File
@@ -10,6 +10,7 @@
- `INT4_SYM` -> `int4`
- `INT4` -> `uint4`
- Add `float8_e4m3fn` and `float8_e5m2` support
- Add quantized matmul support for `float8_e4m3fn`
- Set the default quant mode to `pre`
- Use per token input quant with int8 matmul
- Implement better layer hijacks
@@ -17,6 +18,7 @@
- Fix Conv quant
- Fix lora weight change
- Fix high RAM usage with pre mode
- Fix scale and zero_point not being offloaded
- **IPEX**
- Disabe Dynamic Attention by default on PyTorch 2.7
- Remove GradScaler hijack and use torch.amp.GradScaler instead
+1 -1
View File
@@ -327,7 +327,7 @@ def int8_matmul(
def quantized_linear_forward_fp8_matmul(self, input: torch.FloatTensor) -> torch.FloatTensor:
if input.shape[-1] % 16 != 0 or self.weight.shape[-1] % 16 != 0 or self.weight.shape[-1] % 16 != 0:
if input.shape[-1] % 16 != 0 or self.weight.shape[0] % 16 != 0 or self.weight.shape[1] % 16 != 0:
return torch.nn.functional.linear(input, self.sdnq_decompressor(self.weight, skip_quantized_matmul=True), self.bias)
return fp8_matmul(input, self.weight, self.bias, self.sdnq_decompressor.scale)