diff --git a/CHANGELOG.md b/CHANGELOG.md index 381b83bd0..cc758e77b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,7 +6,10 @@ Feature highlights include: - **ModernUI** has quite some redesign which should make it more user friendly and easier to navigate plus several new UI themes! -- New models [WanAI Wan 2.1](https://wan.video/) for text-to-image workflows, [FreePix F-Lite](https://huggingface.co/Freepik/F-Lite), [Bria 3.2](https://huggingface.co/briaai/BRIA-3.2), [bigASP 2.5](https://civitai.com/models/1789765?modelVersionId=2025412) + If you're still using StandardUI, give [ModernUI](https://vladmandic.github.io/sdnext-docs/Themes/) a try! +- New models: + - [WanAI 2.2](https://wan.video/) in 5B and A14B variants for both *text-to-video* and *image-to-video* workflows as well as *text-to-image* workflow! + - [FreePix F-Lite](https://huggingface.co/Freepik/F-Lite), [Bria 3.2](https://huggingface.co/briaai/BRIA-3.2), [bigASP 2.5](https://civitai.com/models/1789765?modelVersionId=2025412) - Redesigned [LTXVideo](https://vladmandic.github.io/sdnext-docs/Video) interface with support for general video models plus optimized [FramePack](https://vladmandic.github.io/sdnext-docs/FramePack) and [LTXVideo](https://vladmandic.github.io/sdnext-docs/LTX) support - Fully integrated nudity detection and optional censorship with [NudeNet](https://vladmandic.github.io/sdnext-docs/NudeNet) - New background replacement and relightning methods using **Latent Bridge Matching** and new **PixelArt** processing filter @@ -35,8 +38,13 @@ For details, see [ChangeLog](https://github.com/vladmandic/automatic/blob/master - SD.Next [license](https://github.com/vladmandic/sdnext/blob/dev/LICENSE.txt) switched from **aGPL-v3.0** to **Apache-v2.0** this means that SD.Next is now fully compatible with commercial and non-commercial use and redistribution regardless of modifications! - **Models** - - [WanAI Wan 2.1](https://wan.video/) 1.3B and 14B T2V models for T2I workflows - Wan 2.1 is originally designed for *video* workflows, Wan 2.1 model can now also be used for *text-to-image* workflows! + - [WanAI Wan 2.2](https://github.com/Wan-Video/Wan2.2) both 5B and A14B variants, for both T2V and I2V support + go to: *video -> generic -> wan -> pick variant* + optimized support with *VACE*, etc. will follow soon + *note*: quantization and offloading are highly recommended! + - [WanAI Wan](https://wan.video/) T2V models for T2I workflows + Wan is originally designed for *video* workflows, but now also be used for *text-to-image* workflows! + Supports Wan2.1 in 1.3B and 14B variants and Wan2.2 in 5B and A14B variants supports all standard features such as quantization, offloading, TAESD preview generation, LoRA support etc. can also load unet/transformer fine-tunes in safetensors format using UNET loader simply select in *networks -> models -> reference* diff --git a/extensions-builtin/sdnext-modernui b/extensions-builtin/sdnext-modernui index 157bdcac3..ea8de12c2 160000 --- a/extensions-builtin/sdnext-modernui +++ b/extensions-builtin/sdnext-modernui @@ -1 +1 @@ -Subproject commit 157bdcac33b3a480739a2fc475276d6c7b03791b +Subproject commit ea8de12c27aa585bebd0442d789d0877d0fda463 diff --git a/html/reference.json b/html/reference.json index 4561a8eef..8504f822d 100644 --- a/html/reference.json +++ b/html/reference.json @@ -203,6 +203,20 @@ "skip": true, "extras": "sampler: Default" }, + "Wan-AI Wan2.2 5B": { + "path": "Wan-AI/Wan2.2-TI2V-5B-Diffusers", + "preview": "Wan-AI--Wan2.2.jpg", + "desc": "Wan2.2, offering more powerful capabilities, better performance, and superior visual quality. With Wan2.2, we have focused on incorporating the following technical innovations: MoE Architecture, Data Scalling, Cinematic Aesthetics, Efficient High-Definition Hybrid", + "skip": true, + "extras": "sampler: Default" + }, + "Wan-AI Wan2.2 A14B": { + "path": "Wan-AI/Wan2.2-T2V-A14B-Diffusers", + "preview": "Wan-AI--Wan2.2.jpg", + "desc": "Wan2.2, offering more powerful capabilities, better performance, and superior visual quality. With Wan2.2, we have focused on incorporating the following technical innovations: MoE Architecture, Data Scalling, Cinematic Aesthetics, Efficient High-Definition Hybrid", + "skip": true, + "extras": "sampler: Default" + }, "Freepik F-Lite": { "path": "Freepik/F-Lite", @@ -211,6 +225,20 @@ "skip": true, "extras": "sampler: Default, cfg_scale: 3.5" }, + "Freepik F-Lite Texture": { + "path": "Freepik/F-Lite-Texture", + "preview": "Freepik--F-Lite.jpg", + "desc": "F Lite is a 10B parameter diffusion model created by Freepik and Fal, trained exclusively on copyright-safe and SFW content. The model was trained on Freepik's internal dataset comprising approximately 80 million copyright-safe images, making it the first publicly available model of this scale trained exclusively on legally compliant and SFW content.", + "skip": true, + "extras": "sampler: Default, cfg_scale: 3.5" + }, + "Freepik F-Lite 7B": { + "path": "Freepik/F-Lite-7B", + "preview": "Freepik--F-Lite.jpg", + "desc": "F Lite is a 10B parameter diffusion model created by Freepik and Fal, trained exclusively on copyright-safe and SFW content. The model was trained on Freepik's internal dataset comprising approximately 80 million copyright-safe images, making it the first publicly available model of this scale trained exclusively on legally compliant and SFW content.", + "skip": true, + "extras": "sampler: Default, cfg_scale: 3.5" + }, "NVLabs Sana 1.5 1.6B 1k": { "path": "Efficient-Large-Model/SANA1.5_1.6B_1024px_diffusers", diff --git a/models/Reference/Wan-AI--Wan2.2.jpg b/models/Reference/Wan-AI--Wan2.2.jpg new file mode 100644 index 000000000..72e823e5f Binary files /dev/null and b/models/Reference/Wan-AI--Wan2.2.jpg differ diff --git a/modules/lora/network_lora.py b/modules/lora/network_lora.py index 8f06bac6d..fab5b91d4 100644 --- a/modules/lora/network_lora.py +++ b/modules/lora/network_lora.py @@ -3,6 +3,7 @@ import diffusers.models.lora as diffusers_lora import modules.lora.lyco_helpers as lyco_helpers import modules.lora.network as network from modules import devices +from modules.errors import log class ModuleTypeLora(network.ModuleType): @@ -26,8 +27,9 @@ class NetworkModuleLora(network.NetworkModule): if weight is None and none_ok: return None linear_modules = [torch.nn.Linear, torch.nn.modules.linear.NonDynamicallyQuantizableLinear, torch.nn.MultiheadAttention, diffusers_lora.LoRACompatibleLinear] - is_linear = type(self.sd_module) in linear_modules or self.sd_module.__class__.__name__ in {"NNCFLinear", "QLinear", "Linear4bit"} - is_conv = type(self.sd_module) in [torch.nn.Conv2d, diffusers_lora.LoRACompatibleConv] or self.sd_module.__class__.__name__ in {"NNCFConv2d", "QConv2d"} + typ = type(self.sd_module) + is_linear = typ in linear_modules or self.sd_module.__class__.__name__ in ["NNCFLinear", "QLinear", "Linear4bit"] + is_conv = (typ in [torch.nn.Conv2d, diffusers_lora.LoRACompatibleConv]) or (self.sd_module.__class__.__name__ in ["NNCFConv2d", "QConv2d"]) or (typ.__name__ in ['downsampler_block', 'upsampler_block']) if is_linear: weight = weight.reshape(weight.shape[0], -1) module = torch.nn.Linear(weight.shape[1], weight.shape[0], bias=False) @@ -43,7 +45,7 @@ class NetworkModuleLora(network.NetworkModule): elif is_conv and (key == "lora_up.weight" or key == "dyn_down"): module = torch.nn.Conv2d(weight.shape[1], weight.shape[0], (1, 1), bias=False) else: - raise AssertionError(f'Lora unsupported: linear={is_linear} conv={is_conv} key={key} layer={self.network_key} type={type(self.sd_module).__name__}') + raise AssertionError(f'Lora unsupported: key={key} layer={self.network_key} type={typ.__name__}') with torch.no_grad(): if weight.shape != module.weight.shape: weight = weight.reshape(module.weight.shape) diff --git a/modules/video_models/models_def.py b/modules/video_models/models_def.py index 8122fa031..e6f570775 100644 --- a/modules/video_models/models_def.py +++ b/modules/video_models/models_def.py @@ -140,6 +140,30 @@ models = { ], 'WAN Video': [ Model(name='None'), + Model(name='WAN 2.2 5B T2V', + url='https://huggingface.co/Wan-AI/Wan2.2-TI2V-5B-Diffusers', + repo='Wan-AI/Wan2.2-TI2V-5B-Diffusers', + repo_cls=diffusers.WanPipeline, + te_cls=transformers.T5EncoderModel, + dit_cls=diffusers.WanTransformer3DModel), + Model(name='WAN 2.2 5B I2V', + url='https://huggingface.co/Wan-AI/Wan2.2-TI2V-5B-Diffusers', + repo='Wan-AI/Wan2.2-TI2V-5B-Diffusers', + repo_cls=diffusers.WanPipeline, + te_cls=transformers.T5EncoderModel, + dit_cls=diffusers.WanTransformer3DModel), + Model(name='WAN 2.2 A14B T2V', + url='https://huggingface.co/Wan-AI/Wan2.2-T2V-A14B-Diffusers', + repo='Wan-AI/Wan2.2-T2V-A14B-Diffusers', + repo_cls=diffusers.WanPipeline, + te_cls=transformers.T5EncoderModel, + dit_cls=diffusers.WanTransformer3DModel), + Model(name='WAN 2.2 A14B I2V', + url='https://huggingface.co/Wan-AI/Wan2.2-I2V-A14B-Diffusers', + repo='Wan-AI/Wan2.2-T2V-A14B-Diffusers', + repo_cls=diffusers.WanPipeline, + te_cls=transformers.T5EncoderModel, + dit_cls=diffusers.WanTransformer3DModel), Model(name='WAN 2.1 1.3B T2V', url='https://huggingface.co/Wan-AI/Wan2.1-T2V-1.3B-Diffusers', repo='Wan-AI/Wan2.1-T2V-1.3B-Diffusers',