diff --git a/CHANGELOG.md b/CHANGELOG.md index f63bcedbf..bb1903d11 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,9 +5,7 @@ ### Pending - Requires `diffusers==0.30.0` -- [AuraFlow/LavenderFlow](https://github.com/huggingface/diffusers/pull/8796) (previously known as LavenderFlow) -- [Kolors](https://github.com/huggingface/diffusers/pull/8812) -- [ControlNet Union](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0) pipeline +- AuraFlow, Kolors, AlphaVLLM Lumina - FlowMatchHeunDiscreteScheduler enable ### Highlights @@ -15,6 +13,7 @@ Massive update to WiKi with over 20 new pages and articles, now includes guides for nearly all major features Support for new models: - [AlphaVLLM Lumina-Next-SFT](https://huggingface.co/Alpha-VLLM/Lumina-Next-SFT-diffusers) +- [AuraFlow](https://huggingface.co/fal/AuraFlow) - [Kwai Kolors](https://huggingface.co/Kwai-Kolors/Kolors) - [HunyuanDiT 1.2](https://huggingface.co/Tencent-Hunyuan/HunyuanDiT-v1.2-Diffusers) @@ -23,17 +22,21 @@ New **fast-install** mode, new **controlnet-union** *all-in-one* model, support ### New Models +To use and of the new models, simply select model from *Networks -> Reference* and it will be auto-downloaded on first use. + +- [AuraFlow](https://huggingface.co/fal/AuraFlow) + AuraFlow is inspired by SD3 and is by far the largest text-to-image generation model that comes with an Apache 2.0 license + This is a very large model at 6.8B params and nearly 23GB in size, smaller variants are expected in the future + Use scheduler: default or euler flowmatch or heun flowmatch - [AlphaVLLM Lumina-Next-SFT](https://huggingface.co/Alpha-VLLM/Lumina-Next-SFT-diffusers) - to use, simply select from *networks -> reference - use scheduler: default or euler flowmatch or heun flowmatch - note: this model uses T5 XXL variation of text encoder - (previous version of Lumina used Gemma 2B as text encoder) + Lumina-Next-SFT is a Next-DiT model containing 2B parameters, enhanced through high-quality supervised fine-tuning (SFT) + This model uses T5 XXL variation of text encoder (previous version of Lumina used Gemma 2B as text encoder) + Use scheduler: default or euler flowmatch or heun flowmatch - [Kwai Kolors](https://huggingface.co/Kwai-Kolors/Kolors) - to use, simply select from *networks -> reference - note: this is an SDXL style model that replaces standard CLiP-L and CLiP-G text encoders with a massive `chatglm3-6b` encoder - however, this new encoder does support both English and Chinese prompting + Kolors is a large-scale text-to-image generation model based on latent diffusion + This is an SDXL style model that replaces standard CLiP-L and CLiP-G text encoders with a massive `chatglm3-6b` encoder supporting both English and Chinese prompting - [HunyuanDiT 1.2](https://huggingface.co/Tencent-Hunyuan/HunyuanDiT-v1.2-Diffusers) - to use, simply select from *networks -> reference + Hunyuan-DiT is a powerful multi-resolution diffusion transformer (DiT) with fine-grained Chinese understanding ## Update for 2024-07-08 diff --git a/extensions-builtin/sdnext-modernui b/extensions-builtin/sdnext-modernui index 6a570df7a..5e728032c 160000 --- a/extensions-builtin/sdnext-modernui +++ b/extensions-builtin/sdnext-modernui @@ -1 +1 @@ -Subproject commit 6a570df7ada9a048f3ce273851ade9cede9d5c26 +Subproject commit 5e728032c054ce4344d96e327b9e389420711e21 diff --git a/html/reference.json b/html/reference.json index 4d65aa594..6b98e2018 100644 --- a/html/reference.json +++ b/html/reference.json @@ -205,6 +205,14 @@ "extras": "width: 1024, height: 1024" }, + "AuraFlow 0.1": { + "path": "https://huggingface.co/fal/AuraFlow", + "desc": "AuraFlow v0.1, an Open Exploration of Large Rectified Flow Models is inspired by SD3 and is by far the largest text-to-image generation model that comes with an Apache 2.0 license. This model achieves state-of-the-art results on the GenEval benchmark.", + "preview": "fal-AuraFlow.jpg", + "skip": true, + "extras": "width: 1024, height: 1024" + }, + "Kandinsky 2.1": { "path": "kandinsky-community/kandinsky-2-1", "desc": "Kandinsky 2.1 is a text-conditional diffusion model based on unCLIP and latent diffusion, composed of a transformer-based image prior model, a unet diffusion model, and a decoder. Kandinsky 2.1 inherits best practices from Dall-E 2 and Latent diffusion while introducing some new ideas. It uses the CLIP model as a text and image encoder, and diffusion image prior (mapping) between latent spaces of CLIP modalities. This approach increases the visual performance of the model and unveils new horizons in blending images and text-guided image manipulation.", diff --git a/models/Reference/Alpha-VLLM-Lumina-Next-SFT-diffusers.jpg b/models/Reference/Alpha-VLLM-Lumina-Next-SFT-diffusers.jpg index e252bff5b..fb94e2040 100644 Binary files a/models/Reference/Alpha-VLLM-Lumina-Next-SFT-diffusers.jpg and b/models/Reference/Alpha-VLLM-Lumina-Next-SFT-diffusers.jpg differ diff --git a/models/Reference/Kwai-Kolors.jpg b/models/Reference/Kwai-Kolors.jpg index 3ed14d7ce..6d2506926 100644 Binary files a/models/Reference/Kwai-Kolors.jpg and b/models/Reference/Kwai-Kolors.jpg differ diff --git a/models/Reference/fal-AuraFlow.jpg b/models/Reference/fal-AuraFlow.jpg new file mode 100644 index 000000000..bb3baf66d Binary files /dev/null and b/models/Reference/fal-AuraFlow.jpg differ diff --git a/models/Reference/kandinsky-community--kandinsky-2-1.jpg b/models/Reference/kandinsky-community--kandinsky-2-1.jpg index 597c1f865..2fb28d1a4 100644 Binary files a/models/Reference/kandinsky-community--kandinsky-2-1.jpg and b/models/Reference/kandinsky-community--kandinsky-2-1.jpg differ diff --git a/models/Reference/kandinsky-community--kandinsky-3.jpg b/models/Reference/kandinsky-community--kandinsky-3.jpg index 96570921a..f54af2b8c 100644 Binary files a/models/Reference/kandinsky-community--kandinsky-3.jpg and b/models/Reference/kandinsky-community--kandinsky-3.jpg differ diff --git a/models/Reference/playgroundai--playground-v2-256px-base.jpg b/models/Reference/playgroundai--playground-v2-256px-base.jpg index 42a332b91..95645d2f2 100644 Binary files a/models/Reference/playgroundai--playground-v2-256px-base.jpg and b/models/Reference/playgroundai--playground-v2-256px-base.jpg differ diff --git a/models/Reference/playgroundai--playground-v2-512px-base.jpg b/models/Reference/playgroundai--playground-v2-512px-base.jpg index 759945df4..7f03eceb3 100644 Binary files a/models/Reference/playgroundai--playground-v2-512px-base.jpg and b/models/Reference/playgroundai--playground-v2-512px-base.jpg differ diff --git a/models/Reference/segmind--SSD-1B.jpg b/models/Reference/segmind--SSD-1B.jpg index 982a3e4de..2d00260b9 100644 Binary files a/models/Reference/segmind--SSD-1B.jpg and b/models/Reference/segmind--SSD-1B.jpg differ diff --git a/models/Reference/stabilityai--stable-diffusion-3.jpg b/models/Reference/stabilityai--stable-diffusion-3.jpg index da4097d90..6bed3b8be 100644 Binary files a/models/Reference/stabilityai--stable-diffusion-3.jpg and b/models/Reference/stabilityai--stable-diffusion-3.jpg differ diff --git a/modules/model_auraflow.py b/modules/model_auraflow.py new file mode 100644 index 000000000..344e6558a --- /dev/null +++ b/modules/model_auraflow.py @@ -0,0 +1,19 @@ +import torch +import diffusers + + +repo_id = 'fal/AuraFlow' + + +def load_auraflow(_checkpoint_info, diffusers_load_config={}): + from modules import shared, devices + if 'torch_dtype' not in diffusers_load_config: + diffusers_load_config['torch_dtype'] = torch.float16 + + pipe = diffusers.AuraFlowPipeline.from_pretrained( + repo_id, + cache_dir = shared.opts.diffusers_dir, + **diffusers_load_config, + ) + devices.torch_gc() + return pipe diff --git a/modules/sd_models.py b/modules/sd_models.py index 47cc8538f..9109b6a7f 100644 --- a/modules/sd_models.py +++ b/modules/sd_models.py @@ -615,6 +615,8 @@ def detect_pipeline(f: str, op: str = 'model', warning=True, quiet=False): guess = 'Lumina-Next' if 'kolors' in f.lower(): guess = 'Kolors' + if 'auraflow' in f.lower(): + guess = 'AuraFlow' # switch for specific variant if guess == 'Stable Diffusion' and 'inpaint' in f.lower(): guess = 'Stable Diffusion Inpaint' @@ -1014,6 +1016,15 @@ def load_diffuser(checkpoint_info=None, already_loaded_state_dict=None, timer=No if debug_load: errors.display(e, 'Load') return + elif model_type in ['AuraFlow']: # forced pipeline + try: + from modules.model_auraflow import load_auraflow + sd_model = load_auraflow(checkpoint_info, diffusers_load_config) + except Exception as e: + shared.log.error(f'Diffusers Failed loading {op}: {checkpoint_info.path} {e}') + if debug_load: + errors.display(e, 'Load') + return elif model_type in ['Stable Diffusion 3']: try: from modules.model_sd3 import load_sd3 diff --git a/modules/shared_items.py b/modules/shared_items.py index e30bb16aa..478d54cbe 100644 --- a/modules/shared_items.py +++ b/modules/shared_items.py @@ -71,9 +71,10 @@ def get_pipelines(): 'Kandinsky 3': getattr(diffusers, 'Kandinsky3Pipeline', None), 'DeepFloyd IF': getattr(diffusers, 'IFPipeline', None), 'Custom Diffusers Pipeline': getattr(diffusers, 'DiffusionPipeline', None), - 'Kolors': getattr(diffusers, 'StableDiffusionXLPipeline', None), 'InstaFlow': getattr(diffusers, 'StableDiffusionPipeline', None), # dynamically redefined and loaded in sd_models.load_diffuser 'SegMoE': getattr(diffusers, 'StableDiffusionPipeline', None), # dynamically redefined and loaded in sd_models.load_diffuser + 'Kolors': getattr(diffusers, 'KolorsPipeline', None), + 'AuraFlow': getattr(diffusers, 'AuraFlowPipeline', None), } if hasattr(diffusers, 'OnnxStableDiffusionPipeline'): onnx_pipelines = { diff --git a/requirements.txt b/requirements.txt index 69a74cac8..877f4c822 100644 --- a/requirements.txt +++ b/requirements.txt @@ -39,7 +39,7 @@ clip-interrogator==0.6.0 antlr4-python3-runtime==4.9.3 requests==2.31.0 tqdm==4.66.4 -accelerate==0.30.1 +accelerate==0.32.1 opencv-contrib-python-headless==4.9.0.80 einops==0.4.1 gradio==3.43.2 @@ -53,7 +53,7 @@ pandas protobuf==4.25.3 pytorch_lightning==1.9.4 tokenizers==0.19.1 -transformers==4.42.3 +transformers==4.42.4 urllib3==1.26.19 Pillow==10.3.0 timm==0.9.16 diff --git a/wiki b/wiki index 68fa996e9..967fe7a5e 160000 --- a/wiki +++ b/wiki @@ -1 +1 @@ -Subproject commit 68fa996e9231572c244548ef2690adbce018d70b +Subproject commit 967fe7a5ea5117a5eb38578e141df4e74e8e45f8