mirror of
https://github.com/vladmandic/automatic
synced 2026-09-19 17:24:32 +02:00
Update changelog
This commit is contained in:
@@ -10,6 +10,7 @@
|
||||
- `INT4_SYM` -> `int4`
|
||||
- `INT4` -> `uint4`
|
||||
- Add `float8_e4m3fn` and `float8_e5m2` support
|
||||
- Add quantized matmul support for `float8_e4m3fn`
|
||||
- Set the default quant mode to `pre`
|
||||
- Use per token input quant with int8 matmul
|
||||
- Implement better layer hijacks
|
||||
@@ -17,6 +18,7 @@
|
||||
- Fix Conv quant
|
||||
- Fix lora weight change
|
||||
- Fix high RAM usage with pre mode
|
||||
- Fix scale and zero_point not being offloaded
|
||||
- **IPEX**
|
||||
- Disabe Dynamic Attention by default on PyTorch 2.7
|
||||
- Remove GradScaler hijack and use torch.amp.GradScaler instead
|
||||
|
||||
@@ -327,7 +327,7 @@ def int8_matmul(
|
||||
|
||||
|
||||
def quantized_linear_forward_fp8_matmul(self, input: torch.FloatTensor) -> torch.FloatTensor:
|
||||
if input.shape[-1] % 16 != 0 or self.weight.shape[-1] % 16 != 0 or self.weight.shape[-1] % 16 != 0:
|
||||
if input.shape[-1] % 16 != 0 or self.weight.shape[0] % 16 != 0 or self.weight.shape[1] % 16 != 0:
|
||||
return torch.nn.functional.linear(input, self.sdnq_decompressor(self.weight, skip_quantized_matmul=True), self.bias)
|
||||
return fp8_matmul(input, self.weight, self.bias, self.sdnq_decompressor.scale)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user