diff --git a/CHANGELOG.md b/CHANGELOG.md index 0a354cc57..113923d59 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,8 +5,20 @@ - items that require `diffusers==0.27.0.dev`: - EDM samplers for Playground 2.5 - Stable Cascade +- move models to script: + - StabilityAI SVD + - StabilityAI SVD XT 1.0 + - StabilityAI SVD XT 1.1 +- fix reference models: + - Warp Wuerstchen: pipeline does not have all components + - Kandinsky 2.1: pipeline does not have all components + - Kandinsky 2.2: pipeline does not have all components + - StabilityAI Stable Cascade Lite: broken decode in diffusers + - Tsinghua UniDiffuser: no model offload +- untested: + - DeepFloyd IF Medium -## Update for 2024-03-11 +## Update for 2024-03-12 - [Playground v2.5](https://huggingface.co/playgroundai/playground-v2.5-1024px-aesthetic) - new model version from Playground: based on SDXL, but with some cool new concepts @@ -16,7 +28,7 @@ - another very fast & light sdxl model where original unet was compressed and distilled to 54% of original size - download using networks -> reference - *note* to download fp16 variant (recommended), set settings -> diffusers -> preferred model variant - [Stable Cascade](https://github.com/Stability-AI/StableCascade) + [Stable Cascade](https://github.com/Stability-AI/StableCascade) *Full* and *Lite* - large multi-stage high-quality model - download using networks -> reference - see [wiki](https://github.com/vladmandic/automatic/wiki/Stable-Cascade) for details @@ -76,13 +88,19 @@ - see *settings -> image options -> watermarking* - invisible watermark: using steganogephy - image watermark: overlaid on top of image +- **Reference models** + - additional reference models available for single-click download & run: + *Stable Cascade, Stable Cascade lite, Stable Video Diffusion XT 1.1* + - reference models will now download *fp16* variation by default + - reference models will print recommended settings to log if present + - new setting in extra network: *use reference values when available* + disabled by default, if enabled will force use of reference settings for models that have them - **Improvements** - **FaceID** extend support for LoRA, HyperTile and FreeU, thanks @Trojaner - **Tiling** now extends to both Unet and VAE producing smoother outputs, thanks @AI-Casanova - new setting in image options: *include mask in output* - improved params parsing from from prompt string and styles - default theme updates and additional built-in theme *black-gray* - - new setting in extra network: "use reference values when available" - support models with their own YAML model config files - support models with their own JSON per-component config files, for example: `playground-v2.5_vae.config` - **ROCm** diff --git a/html/reference.json b/html/reference.json index 0e38c9760..ba53a7b5a 100644 --- a/html/reference.json +++ b/html/reference.json @@ -1,214 +1,251 @@ { - "DreamShaper SD 1.5 v8": { + + "DreamShaper SD v8": { + "original": true, "path": "dreamshaper_8.safetensors@https://civitai.com/api/download/models/128713", - "desc": "Showcase finetuned model based on Stable diffusion 1.5", "preview": "dreamshaper_8.jpg", - "extras": "width: 768, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "original": true - }, - "DreamShaper SD XL Turbo": { - "path": "dreamshaperXL_turboDpmppSDE.safetensors@https://civitai.com/api/download/models/251662", - "desc": "Showcase finetuned model based on Stable diffusion XL", - "extras": "width: 1024, height: 1024, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "preview": "dreamshaperXL_turboDpmppSDE.jpg" - }, - "Juggernaut Reborn": { - "path": "juggernaut_reborn.safetensors@https://civitai.com/api/download/models/274039", "desc": "Showcase finetuned model based on Stable diffusion 1.5", - "preview": "juggernaut_reborn.jpg", - "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "original": true + "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0" }, - "Juggernaut XL v7 RunDiffusion": { - "path": "juggernautXL_v7Rundiffusion.safetensors@https://civitai.com/api/download/models/240840", + "Dreamshaper SD v7 LCM": { + "path": "SimianLuo/LCM_Dreamshaper_v7", + "preview": "SimianLuo--LCM_Dreamshaper_v7.jpg", + "desc": "Latent Consistencey Models enable swift inference with minimal steps on any pre-trained LDMs, including Stable Diffusion. By distilling classifier-free guidance into the model's input, LCM can generate high-quality images in very short inference time. LCM can generate quality images in as few as 3-4 steps, making it blazingly fast.", + "extras": "width: 512, height: 512, sampler: LCM, steps: 4, cfg_scale: 0.0" + }, + "DreamShaper SD-XL Turbo": { + "path": "dreamshaperXL_v21TurboDPMSDE.safetensors@https://civitai.com/api/download/models/351306", + "preview": "dreamshaperXL_v21TurboDPMSDE.jpg", "desc": "Showcase finetuned model based on Stable diffusion XL", - "extras": "width: 1024, height: 1024, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "preview": "juggernautXL_v7Rundiffusion.jpg" + "extras": "width: 1024, height: 1024, sampler: DPM SDE, steps: 8, cfg_scale: 2.0" + }, + "Juggernaut SD Reborn": { + "original": true, + "path": "juggernaut_reborn.safetensors@https://civitai.com/api/download/models/274039", + "preview": "juggernaut_reborn.jpg", + "desc": "Showcase finetuned model based on Stable diffusion 1.5", + "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0" }, + "Juggernaut SD-XL v9": { + "path": "juggernautXL_v9Rundiffusionphoto2.safetensors@https://civitai.com/api/download/models/348913", + "preview": "juggernautXL_v9Rundiffusionphoto2.jpg", + "desc": "Showcase finetuned model based on Stable diffusion XL", + "extras": "width: 1024, height: 1024, sampler: DEIS, steps: 20, cfg_scale: 6.0" + }, + "Juggernaut SD-XL v9 Lightning": { + "path": "juggernautXL_v9Rdphoto2Lightning.safetensors@https://civitai.com/api/download/models/357609", + "preview": "juggernautXL_v9Rdphoto2Lightning.jpg", + "desc": "Showcase finetuned model based on Stable diffusion XL", + "extras": "width: 1024, height: 1024, sampler: DPM SDE, steps: 6, cfg_scale: 2.0" + }, + "Tempest SD-XL v0.1": { + "path": "TempestV0.1-Artistic.safetensors@https://huggingface.co/dataautogpt3/TempestV0.1/resolve/main/TempestV0.1-Artistic.safetensors?download=true", + "preview": "TempestV0.1-Artistic.jpg", + "desc": "The TempestV0.1 Initiative is a powerhouse in image generation, leveraging an unparalleled dataset of over 6 million images. The collection's vast scale, with resolutions from 1400x2100 to 4800x7200, encompasses 200GB of high-quality content.", + "extras": "width: 2048, height: 1024, sampler: DEIS, steps: 40, cfg_scale: 6.0" + }, + "RunwayML SD 1.5": { - "path": "runwayml/stable-diffusion-v1-5", - "alt": "v1-5-pruned-emaonly.safetensors@https://huggingface.co/runwayml/stable-diffusion-v1-5/resolve/main/v1-5-pruned-emaonly.safetensors?download=true", + "original": true, + "path": "v1-5-pruned-fp16-emaonly.safetensors@https://huggingface.co/Aptronym/SDNext/resolve/main/Reference/v1-5-pruned-fp16-emaonly.safetensors?download=true", + "preview": "v1-5-pruned-fp16-emaonly.jpg", "desc": "Stable Diffusion 1.5 is the base model all other 1.5 checkpoint were trained from. It's a latent text-to-image diffusion model capable of generating photo-realistic images given any text input. The Stable-Diffusion-v1-5 checkpoint was initialized with the weights of the Stable-Diffusion-v1-2 checkpoint and subsequently fine-tuned on 595k steps at resolution 512x512.", - "preview": "runwayml--stable-diffusion-v1-5.jpg", - "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "original": true + "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0" }, - "StabilityAI SD 2.1 EMA": { - "path": "stabilityai/stable-diffusion-2-1-base", - "alt": "v2-1_512-ema-pruned.safetensors@https://huggingface.co/stabilityai/stable-diffusion-2-1-base/resolve/main/v2-1_512-ema-pruned.safetensors?download=true", - "desc": "This stable-diffusion-2-1-base model fine-tunes stable-diffusion-2-base (512-base-ema.ckpt) with 220k extra steps taken", - "preview": "stabilityai--stable-diffusion-2.1-base.jpg", - "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "original": true - }, - "StabilityAI SD 2.1 V": { - "path": "stabilityai/stable-diffusion-2-1-base", - "alt": "v2-1_768-ema-pruned.safetensors@https://huggingface.co/stabilityai/stable-diffusion-2-1/resolve/main/v2-1_768-ema-pruned.safetensors?download=true", - "desc": "This stable-diffusion-2 model is resumed from stable-diffusion-2-base (512-base-ema.ckpt) and trained for 150k steps using a v-objective on the same dataset. Resumed for another 140k steps on 768x768 images", - "preview": "stabilityai--stable-diffusion-2.1-base.jpg", - "extras": "width: 768, height: 768, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "original": true - }, - "StabilityAI SD-XL 1.0 Base": { - "path": "huggingface/stabilityai/stable-diffusion-xl-base-1.0", + "StabilityAI SD 2.1": { + "path": "huggingface/stabilityai/stable-diffusion-2-1-base", + "preview": "stabilityai--stable-diffusion-2-1-base.jpg", "skip": true, "variant": "fp16", + "desc": "This stable-diffusion-2-1-base model fine-tunes stable-diffusion-2-base (512-base-ema.ckpt) with 220k extra steps taken", + "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0" + }, + "StabilityAI SD 2.1 V": { + "path": "huggingface/stabilityai/stable-diffusion-2-1", + "preview": "stabilityai--stable-diffusion-2-1.jpg", + "skip": true, + "variant": "fp16", + "desc": "This stable-diffusion-2 model is resumed from stable-diffusion-2-base (512-base-ema.ckpt) and trained for 150k steps using a v-objective on the same dataset. Resumed for another 140k steps on 768x768 images", + "extras": "width: 768, height: 768, sampler: DEIS, steps: 20, cfg_scale: 6.0" + }, + "StabilityAI SD-XL 1.0 Base": { + "path": "sd_xl_base_1.0.safetensors@https://huggingface.co/stabilityai/stable-diffusion-xl-base-1.0/resolve/main/sd_xl_base_1.0.safetensors?download=true", + "preview": "sd_xl_base_1.0.jpg", "desc": "Stable Diffusion XL (SDXL) is the latest AI image generation model that is tailored towards more photorealistic outputs with more detailed imagery and composition compared to previous SD models, including SD 2.1. It can make realistic faces, legible text within the images, and better image composition, all while using shorter and simpler prompts at a greatly increased base resolution of 1024x1024. Just like its predecessors, SDXL has the ability to generate image variations using image-to-image prompting, inpainting (reimagining of the selected parts of an image), and outpainting (creating new parts that lie outside the image borders).", - "extras": "width: 1024, height: 1024, sampler: DEIS, steps: 20, cfg_scale: 6.0", - "preview": "stabilityai--stable-diffusion-xl-base-1.0.jpg" - }, - "StabilityAI SD 2.1 Turbo": { - "_path": "stabilityai/sd-turbo", - "path": "sd_turbo.safetensors@https://huggingface.co/stabilityai/sd-turbo/resolve/main/sd_turbo.safetensors?download=true", - "variant": "fp16", - "desc": "SD-Turbo is a distilled version of Stable Diffusion 2.1, trained for real-time synthesis. SD-Turbo is based on a novel training method called Adversarial Diffusion Distillation (ADD) (see the technical report), which allows sampling large-scale foundational image diffusion models in 1 to 4 steps at high image quality. This approach uses score distillation to leverage large-scale off-the-shelf image diffusion models as a teacher signal and combines this with an adversarial loss to ensure high image fidelity even in the low-step regime of one or two sampling steps.", - "preview": "stabilityai--sd-turbo.jpg", - "original": true - }, - "StabilityAI SD-XL Turbo": { - "_path": "stabilityai/sdxl-turbo", - "path": "sdxl_turbo.safetensors@https://huggingface.co/stabilityai/sdxl-turbo/resolve/main/sd_xl_turbo_1.0_fp16.safetensors?download=true", - "variant": "fp16", - "desc": "SDXL-Turbo is a distilled version of SDXL 1.0, trained for real-time synthesis. SDXL-Turbo is based on a novel training method called Adversarial Diffusion Distillation (ADD) (see the technical report), which allows sampling large-scale foundational image diffusion models in 1 to 4 steps at high image quality. This approach uses score distillation to leverage large-scale off-the-shelf image diffusion models as a teacher signal and combines this with an adversarial loss to ensure high image fidelity even in the low-step regime of one or two sampling steps.", - "preview": "stabilityai--sdxl-turbo.jpg" - }, - "StabilityAI Stable Video Diffusion": { - "path": "stabilityai/stable-video-diffusion-img2vid", - "desc": "(SVD) Image-to-Video is a latent diffusion model trained to generate short video clips from an image conditioning. This model was trained to generate 14 frames at resolution 576x1024 given a context frame of the same size. We also finetune the widely used f8-decoder for temporal consistency.", - "preview": "stabilityai--stable-video-diffusion-img2vid.jpg" - }, - "StabilityAI Stable Video Diffusion XT": { - "path": "stabilityai/stable-video-diffusion-img2vid-xt", - "desc": "(SVD) Image-to-Video is a latent diffusion model trained to generate short video clips from an image conditioning. This model was trained to generate 25 frames at resolution 576x1024 given a context frame of the same size, finetuned from SVD Image-to-Video [14 frames]. We also finetune the widely used f8-decoder for temporal consistency.", - "preview": "stabilityai--stable-video-diffusion-img2vid-xt.jpg" + "extras": "width: 1024, height: 1024, sampler: DEIS, steps: 20, cfg_scale: 6.0" }, + "StabilityAI Stable Cascade": { "path": "huggingface/stabilityai/stable-cascade", "skip": true, + "variant": "bf16", "desc": "Stable Cascade is a diffusion model built upon the Würstchen architecture and its main difference to other models like Stable Diffusion is that it is working at a much smaller latent space. Why is this important? The smaller the latent space, the faster you can run inference and the cheaper the training becomes. How small is the latent space? Stable Diffusion uses a compression factor of 8, resulting in a 1024x1024 image being encoded to 128x128. Stable Cascade achieves a compression factor of 42, meaning that it is possible to encode a 1024x1024 image to 24x24, while maintaining crisp reconstructions. The text-conditional model is then trained in the highly compressed latent space. Previous versions of this architecture, achieved a 16x cost reduction over Stable Diffusion 1.5", - "preview": "stabilityai--stable-cascade.jpg" + "preview": "stabilityai--stable-cascade.jpg", + "extras": "width: 1024, height: 1024, sampler: Default, cfg_scale: 4.0, image_cfg_scale: 1.0" }, + "StabilityAI Stable Cascade Lite": { + "path": "huggingface/stabilityai/stable-cascade-lite", + "skip": true, + "variant": "bf16", + "desc": "Stable Cascade is a diffusion model built upon the Würstchen architecture and its main difference to other models like Stable Diffusion is that it is working at a much smaller latent space. Why is this important? The smaller the latent space, the faster you can run inference and the cheaper the training becomes. How small is the latent space? Stable Diffusion uses a compression factor of 8, resulting in a 1024x1024 image being encoded to 128x128. Stable Cascade achieves a compression factor of 42, meaning that it is possible to encode a 1024x1024 image to 24x24, while maintaining crisp reconstructions. The text-conditional model is then trained in the highly compressed latent space. Previous versions of this architecture, achieved a 16x cost reduction over Stable Diffusion 1.5", + "preview": "stabilityai--stable-cascade.jpg", + "extras": "width: 1024, height: 1024, sampler: Default, cfg_scale: 4.0, image_cfg_scale: 1.0" + }, + + "StabilityAI SVD": { + "path": "stabilityai/stable-video-diffusion-img2vid", + "preview": "stabilityai--stable-video-diffusion-img2vid.jpg", + "desc": "(SVD) Image-to-Video is a latent diffusion model trained to generate short video clips from an image conditioning. This model was trained to generate 14 frames at resolution 576x1024 given a context frame of the same size. We also finetune the widely used f8-decoder for temporal consistency.", + "variant": "fp16", + "extras": "width: 1024, height: 576, sampler: Default, steps: 20" + }, + "StabilityAI SVD XT 1.0": { + "path": "stabilityai/stable-video-diffusion-img2vid-xt", + "preview": "stabilityai--stable-video-diffusion-img2vid-xt.jpg", + "desc": "(SVD) Image-to-Video is a latent diffusion model trained to generate short video clips from an image conditioning. This model was trained to generate 25 frames at resolution 576x1024 given a context frame of the same size, finetuned from SVD Image-to-Video [14 frames]. We also finetune the widely used f8-decoder for temporal consistency.", + "variant": "fp16", + "extras": "width: 1024, height: 576, sampler: Default, steps: 20" + }, + "StabilityAI SVD XT 1.1": { + "path": "stabilityai/stable-video-diffusion-img2vid-xt-1-1", + "preview": "stabilityai--stable-video-diffusion-img2vid-xt.jpg", + "desc": "(SVD 1.1) Image-to-Video is a latent diffusion model trained to generate short video clips from an image conditioning. This model was trained to generate 25 frames at resolution 1024x576 given a context frame of the same size, finetuned from SVD Image-to-Video [25 frames].", + "variant": "fp16", + "extras": "width: 1024, height: 576, sampler: Default, steps: 20" + }, + "Segmind Vega": { - "path": "segmind/Segmind-Vega", + "path": "huggingface/segmind/Segmind-Vega", + "preview": "segmind--Segmind-Vega.jpg", "desc": "The Segmind-Vega Model is a distilled version of the Stable Diffusion XL (SDXL), offering a remarkable 70% reduction in size and an impressive 100% speedup while retaining high-quality text-to-image generation capabilities. Trained on diverse datasets, including Grit and Midjourney scrape data, it excels at creating a wide range of visual content based on textual prompts. Employing a knowledge distillation strategy, Segmind-Vega leverages the teachings of several expert models, including SDXL, ZavyChromaXL, and JuggernautXL, to combine their strengths and produce compelling visual outputs.", - "preview": "segmind--Segmind-Vega.jpg" + "variant": "fp16", + "skip": true, + "extras": "width: 1024, height: 1024, sampler: Default, cfg_scale: 9.0" }, "Segmind SSD-1B": { - "path": "segmind/SSD-1B", + "path": "huggingface/segmind/SSD-1B", + "preview": "segmind--SSD-1B.jpg", "desc": "The Segmind Stable Diffusion Model (SSD-1B) offers a compact, efficient, and distilled version of the SDXL model. At 50% smaller and 60% faster than Stable Diffusion XL (SDXL), it provides quick and seamless performance without sacrificing image quality.", - "preview": "segmind--SSD-1B.jpg" + "variant": "fp16", + "skip": true, + "extras": "width: 1024, height: 1024, sampler: Default, cfg_scale: 9.0" }, "Segmind Tiny": { "path": "segmind/tiny-sd", + "preview": "segmind--tiny-sd.jpg", "desc": "Segmind's Tiny-SD offers a compact, efficient, and distilled version of Realistic Vision 4.0 and is up to 80% faster than SD1.5", - "preview": "segmind--tiny-sd.jpg" + "extras": "width: 512, height: 512, sampler: Default, cfg_scale: 9.0" }, "Segmind SegMoE SD 4x2": { "path": "segmind/SegMoE-SD-4x2-v0", + "preview": "segmind--SegMoE-SD-4x2-v0.jpg", "desc": "SegMoE-SD-4x2-v0 is an untrained Segmind Mixture of Diffusion Experts Model generated using segmoe from 4 Expert SD1.5 models. SegMoE is a powerful framework for dynamically combining Stable Diffusion Models into a Mixture of Experts within minutes without training", - "preview": "segmind--SegMoE-SD-4x2-v0.jpg" - }, - "Segmind SegMoE XL 2x1": { - "path": "segmind/SegMoE-2x1-v0", - "desc": "SegMoE-2x1-v0 is an untrained Segmind Mixture of Diffusion Experts Model generated using segmoe from 2 Expert SDXL models. SegMoE is a powerful framework for dynamically combining Stable Diffusion Models into a Mixture of Experts within minutes without training", - "preview": "segmind--SegMoE-2x1-v0.jpg" + "extras": "width: 512, height: 512, sampler: Default" }, "Segmind SegMoE XL 4x2": { "path": "segmind/SegMoE-4x2-v0", + "preview": "segmind--SegMoE-4x2-v0.jpg", "desc": "SegMoE-4x2-v0 is an untrained Segmind Mixture of Diffusion Experts Model generated using segmoe from 4 Expert SDXL models. SegMoE is a powerful framework for dynamically combining Stable Diffusion Models into a Mixture of Experts within minutes without training", - "preview": "segmind--SegMoE-4x2-v0.jpg" + "extras": "width: 1024, height: 1024, sampler: Default" }, - "LCM SD-1.5 Dreamshaper 7": { - "path": "SimianLuo/LCM_Dreamshaper_v7", - "desc": "Latent Consistencey Models enable swift inference with minimal steps on any pre-trained LDMs, including Stable Diffusion. By distilling classifier-free guidance into the model's input, LCM can generate high-quality images in very short inference time. LCM can generate quality images in as few as 3-4 steps, making it blazingly fast.", - "preview": "SimianLuo--LCM_Dreamshaper_v7.jpg" - }, - "Pixart-α XL 2 Medium 512": { + + "Pixart-α XL 2 Medium": { "path": "PixArt-alpha/PixArt-XL-2-512x512", "desc": "PixArt-α is a Transformer-based T2I diffusion model whose image generation quality is competitive with state-of-the-art image generators (e.g., Imagen, SDXL, and even Midjourney), and the training speed markedly surpasses existing large-scale T2I models. Extensive experiments demonstrate that PIXART-α excels in image quality, artistry, and semantic control. It can directly generate 512px images from text prompts within a single sampling process.", - "preview": "PixArt-alpha--PixArt-XL-2-512x512.jpg" + "preview": "PixArt-alpha--PixArt-XL-2-512x512.jpg", + "extras": "width: 512, height: 512, sampler: Default, cfg_scale: 2.0" }, - "Pixart-α XL 2 Large 1024": { + "Pixart-α XL 2 Large": { "path": "PixArt-alpha/PixArt-XL-2-1024-MS", "desc": "PixArt-α is a Transformer-based T2I diffusion model whose image generation quality is competitive with state-of-the-art image generators (e.g., Imagen, SDXL, and even Midjourney), and the training speed markedly surpasses existing large-scale T2I models. Extensive experiments demonstrate that PIXART-α excels in image quality, artistry, and semantic control. It can directly generate 1024px images from text prompts within a single sampling process.", - "preview": "PixArt-alpha--PixArt-XL-2-1024-MS.jpg" - }, - "Pixart-α XL 2 Large LCM": { - "path": "PixArt-alpha/PixArt-LCM-XL-2-1024-MS", - "desc": "Pixart-α consists of pure transformer blocks for latent diffusion: It can directly generate 1024px images from text prompts within a single sampling process. LCMs is a diffusion distillation method which predict PF-ODE's solution directly in latent space, achieving super fast inference with few steps. Following LCM LoRA, we illustrative of the generation speed we achieve on various computers. Let us stress again how liberating it is to explore image generation so easily with PixArt-LCM.", - "preview": "PixArt-alpha--PixArt-XL-2-1024-MS.jpg" - }, - "Warp Wuerstchen": { - "path": "warp-ai/wuerstchen", - "desc": "Würstchen is a diffusion model whose text-conditional model works in a highly compressed latent space of images. Why is this important? Compressing data can reduce computational costs for both training and inference by magnitudes. Training on 1024x1024 images, is way more expensive than training at 32x32. Usually, other works make use of a relatively small compression, in the range of 4x - 8x spatial compression. Würstchen takes this to an extreme. Through its novel design, we achieve a 42x spatial compression. Würstchen employs a two-stage compression, what we call Stage A and Stage B. Stage A is a VQGAN, and Stage B is a Diffusion Autoencoder (more details can be found in the paper). A third model, Stage C, is learned in that highly compressed latent space. This training requires fractions of the compute used for current top-performing models, allowing also cheaper and faster inference.", - "preview": "warp-ai--wuerstchen.jpg" + "preview": "PixArt-alpha--PixArt-XL-2-1024-MS.jpg", + "extras": "width: 1024, height: 1024, sampler: Default, cfg_scale: 2.0" }, + "Kandinsky 2.1": { "path": "kandinsky-community/kandinsky-2-1", "desc": "Kandinsky 2.1 is a text-conditional diffusion model based on unCLIP and latent diffusion, composed of a transformer-based image prior model, a unet diffusion model, and a decoder. Kandinsky 2.1 inherits best practices from Dall-E 2 and Latent diffusion while introducing some new ideas. It uses the CLIP model as a text and image encoder, and diffusion image prior (mapping) between latent spaces of CLIP modalities. This approach increases the visual performance of the model and unveils new horizons in blending images and text-guided image manipulation.", - "preview": "kandinsky-community--kandinsky-2-1.jpg" + "preview": "kandinsky-community--kandinsky-2-1.jpg", + "extras": "width: 768, height: 768, sampler: Default" }, "Kandinsky 2.2": { "path": "kandinsky-community/kandinsky-2-2-decoder", "desc": "Kandinsky 2.2 is a text-conditional diffusion model (+0.1!) based on unCLIP and latent diffusion, composed of a transformer-based image prior model, a unet diffusion model, and a decoder. Kandinsky 2.1 inherits best practices from Dall-E 2 and Latent diffusion while introducing some new ideas. It uses the CLIP model as a text and image encoder, and diffusion image prior (mapping) between latent spaces of CLIP modalities. This approach increases the visual performance of the model and unveils new horizons in blending images and text-guided image manipulation.", - "preview": "kandinsky-community--kandinsky-2-2-decoder.jpg" + "preview": "kandinsky-community--kandinsky-2-2-decoder.jpg", + "extras": "width: 768, height: 768, sampler: Default" }, "Kandinsky 3": { "path": "kandinsky-community/kandinsky-3", "desc": "Kandinsky 3.0 is an open-source text-to-image diffusion model built upon the Kandinsky2-x model family. In comparison to its predecessors, Kandinsky 3.0 incorporates more data and specifically related to Russian culture, which allows to generate pictures related to Russin culture. Furthermore, enhancements have been made to the text understanding and visual quality of the model, achieved by increasing the size of the text encoder and Diffusion U-Net models, respectively.", - "preview": "kandinsky-community--kandinsky-3.jpg" + "preview": "kandinsky-community--kandinsky-3.jpg", + "variant": "fp16", + "extras": "width: 1024, height: 1024, sampler: Default" }, + "Playground v1": { "path": "playgroundai/playground-v1", "desc": "Playground v1 is a latent diffusion model that improves the overall HDR quality to get more stunning images.", - "preview": "playgroundai--playground-v1.jpg" + "preview": "playgroundai--playground-v1.jpg", + "extras": "width: 512, height: 512, sampler: Default" }, - "Playground v2 256": { + "Playground v2 Small": { "path": "playgroundai/playground-v2-256px-base", "desc": "Playground v2 is a diffusion-based text-to-image generative model. The model was trained from scratch by the research team at Playground. Images generated by Playground v2 are favored 2.5 times more than those produced by Stable Diffusion XL, according to Playground’s user study.", - "preview": "playgroundai--playground-v2-256px-base.jpg" + "preview": "playgroundai--playground-v2-256px-base.jpg", + "extras": "width: 256, height: 256, sampler: Default" }, - "Playground v2 512": { + "Playground v2 Medium": { "path": "playgroundai/playground-v2-512px-base", "desc": "Playground v2 is a diffusion-based text-to-image generative model. The model was trained from scratch by the research team at Playground. Images generated by Playground v2 are favored 2.5 times more than those produced by Stable Diffusion XL, according to Playground’s user study.", - "preview": "playgroundai--playground-v2-512px-base.jpg" + "preview": "playgroundai--playground-v2-512px-base.jpg", + "extras": "width: 512, height: 512, sampler: Default" }, - "Playground v2 1024": { + "Playground v2 Large": { "path": "playgroundai/playground-v2-1024px-aesthetic", "desc": "Playground v2 is a diffusion-based text-to-image generative model. The model was trained from scratch by the research team at Playground. Images generated by Playground v2 are favored 2.5 times more than those produced by Stable Diffusion XL, according to Playground’s user study.", - "preview": "playgroundai--playground-v2-1024px-aesthetic.jpg" + "preview": "playgroundai--playground-v2-1024px-aesthetic.jpg", + "extras": "width: 1024, height: 1024, sampler: Default" }, "Playground v2.5": { "path": "playground-v2.5-1024px-aesthetic.fp16.safetensors@https://huggingface.co/playgroundai/playground-v2.5-1024px-aesthetic/resolve/main/playground-v2.5-1024px-aesthetic.fp16.safetensors?download=true", "desc": "Playground v2.5 is a diffusion-based text-to-image generative model, and a successor to Playground v2. Playground v2.5 is the state-of-the-art open-source model in aesthetic quality. Our user studies demonstrate that our model outperforms SDXL, Playground v2, PixArt-α, DALL-E 3, and Midjourney 5.2.", - "preview": "playgroundai--playground-v2-1024px-aesthetic.jpg" - }, - "DeepFloyd IF Medium": { - "path": "DeepFloyd/IF-I-M-v1.0", - "desc": "DeepFloyd-IF is a pixel-based text-to-image triple-cascaded diffusion model, that can generate pictures with new state-of-the-art for photorealism and language understanding. The result is a highly efficient model that outperforms current state-of-the-art models, achieving a zero-shot FID-30K score of 6.66 on the COCO dataset. It is modular and composed of frozen text mode and three pixel cascaded diffusion modules, each designed to generate images of increasing resolution: 64x64, 256x256, and 1024x1024.", - "preview": "DeepFloyd--IF-I-M-v1.0.jpg" + "preview": "playgroundai--playground-v2-1024px-aesthetic.jpg", + "extras": "width: 1024, height: 1024, sampler: DPM++ 2M EDM" }, + "aMUSEd 256": { - "path": "amused/amused-256", + "path": "huggingface/amused/amused-256", + "skip": true, "desc": "Amused is a lightweight text to image model based off of the muse architecture. Amused is particularly useful in applications that require a lightweight and fast model such as generating many images quickly at once.", - "preview": "amused--amused-256.jpg" + "preview": "amused--amused-256.jpg", + "extras": "width: 256, height: 256, sampler: Default" }, "aMUSEd 512": { "path": "amused/amused-512", "desc": "Amused is a lightweight text to image model based off of the muse architecture. Amused is particularly useful in applications that require a lightweight and fast model such as generating many images quickly at once.", - "preview": "amused--amused-512.jpg" + "preview": "amused--amused-512.jpg", + "extras": "width: 512, height: 512, sampler: Default" + }, + + "Warp Wuerstchen": { + "path": "warp-ai/wuerstchen", + "desc": "Würstchen is a diffusion model whose text-conditional model works in a highly compressed latent space of images. Why is this important? Compressing data can reduce computational costs for both training and inference by magnitudes. Training on 1024x1024 images, is way more expensive than training at 32x32. Usually, other works make use of a relatively small compression, in the range of 4x - 8x spatial compression. Würstchen takes this to an extreme. Through its novel design, we achieve a 42x spatial compression. Würstchen employs a two-stage compression, what we call Stage A and Stage B. Stage A is a VQGAN, and Stage B is a Diffusion Autoencoder (more details can be found in the paper). A third model, Stage C, is learned in that highly compressed latent space. This training requires fractions of the compute used for current top-performing models, allowing also cheaper and faster inference.", + "preview": "warp-ai--wuerstchen.jpg", + "extras": "width: 1024, height: 1024, sampler: Default, cfg_scale: 4.0, image_cfg_scale: 0.0" }, "KOALA 700M": { "path": "huggingface/etri-vilab/koala-700m-llava-cap", "variant": "fp16", "skip": true, "desc": "Fast text-to-image model, called KOALA, by compressing SDXL's U-Net and distilling knowledge from SDXL into our model. KOALA-700M can generate a 1024x1024 image in less than 1.5 seconds on an NVIDIA 4090 GPU, which is more than 2x faster than SDXL.", - "preview": "etri-vilab--koala-700m-llava-cap.jpg" + "preview": "etri-vilab--koala-700m-llava-cap.jpg", + "extras": "width: 1024, height: 1024, sampler: Default" }, "Tsinghua UniDiffuser": { "path": "thu-ml/unidiffuser-v1", "desc": "UniDiffuser is a unified diffusion framework to fit all distributions relevant to a set of multi-modal data in one transformer. UniDiffuser is able to perform image, text, text-to-image, image-to-text, and image-text pair generation by setting proper timesteps without additional overhead.\nSpecifically, UniDiffuser employs a variation of transformer, called U-ViT, which parameterizes the joint noise prediction network. Other components perform as encoders and decoders of different modalities, including a pretrained image autoencoder from Stable Diffusion, a pretrained image ViT-B/32 CLIP encoder, a pretrained text ViT-L CLIP encoder, and a GPT-2 text decoder finetuned by ourselves.", - "preview": "thu-ml--unidiffuser-v1.jpg" + "preview": "thu-ml--unidiffuser-v1.jpg", + "extras": "width: 512, height: 512, sampler: Default" }, "SalesForce BLIP-Diffusion": { "path": "salesforce/blipdiffusion", @@ -219,5 +256,12 @@ "path": "XCLiu/instaflow_0_9B_from_sd_1_5", "desc": "InstaFlow is an ultra-fast, one-step image generator that achieves image quality close to Stable Diffusion. This efficiency is made possible through a recent Rectified Flow technique, which trains probability flows with straight trajectories, hence inherently requiring only a single step for fast inference.", "preview": "XCLiu--instaflow_0_9B_from_sd_1_5.jpg" + }, + "DeepFloyd IF Medium": { + "path": "DeepFloyd/IF-I-M-v1.0", + "desc": "DeepFloyd-IF is a pixel-based text-to-image triple-cascaded diffusion model, that can generate pictures with new state-of-the-art for photorealism and language understanding. The result is a highly efficient model that outperforms current state-of-the-art models, achieving a zero-shot FID-30K score of 6.66 on the COCO dataset. It is modular and composed of frozen text mode and three pixel cascaded diffusion modules, each designed to generate images of increasing resolution: 64x64, 256x256, and 1024x1024.", + "preview": "DeepFloyd--IF-I-M-v1.0.jpg", + "extras": "width: 1024, height: 1024, sampler: Default" } -} \ No newline at end of file + +} diff --git a/models/Reference/TempestV0.1-Artistic.jpg b/models/Reference/TempestV0.1-Artistic.jpg new file mode 100644 index 000000000..ab045a5ca Binary files /dev/null and b/models/Reference/TempestV0.1-Artistic.jpg differ diff --git a/models/Reference/dreamshaperXL_turboDpmppSDE.jpg b/models/Reference/dreamshaperXL_v21TurboDPMSDE.jpg similarity index 100% rename from models/Reference/dreamshaperXL_turboDpmppSDE.jpg rename to models/Reference/dreamshaperXL_v21TurboDPMSDE.jpg diff --git a/models/Reference/juggernautXL_v7Rundiffusion.jpg b/models/Reference/juggernautXL_v9Rdphoto2Lightning.jpg similarity index 100% rename from models/Reference/juggernautXL_v7Rundiffusion.jpg rename to models/Reference/juggernautXL_v9Rdphoto2Lightning.jpg diff --git a/models/Reference/juggernautXL_v9Rundiffusionphoto2.jpg b/models/Reference/juggernautXL_v9Rundiffusionphoto2.jpg new file mode 100644 index 000000000..cbce7cb32 Binary files /dev/null and b/models/Reference/juggernautXL_v9Rundiffusionphoto2.jpg differ diff --git a/models/Reference/stabilityai--stable-diffusion-xl-base-1.0.jpg b/models/Reference/sd_xl_base_1.0.jpg similarity index 100% rename from models/Reference/stabilityai--stable-diffusion-xl-base-1.0.jpg rename to models/Reference/sd_xl_base_1.0.jpg diff --git a/models/Reference/stabilityai--sdxl-turbo.jpg b/models/Reference/sdxl_turbo.jpg similarity index 100% rename from models/Reference/stabilityai--sdxl-turbo.jpg rename to models/Reference/sdxl_turbo.jpg diff --git a/models/Reference/stabilityai--sd-turbo.jpg b/models/Reference/stabilityai--stable-diffusion-2-1-base.jpg similarity index 100% rename from models/Reference/stabilityai--sd-turbo.jpg rename to models/Reference/stabilityai--stable-diffusion-2-1-base.jpg diff --git a/models/Reference/stabilityai--stable-diffusion-2.1-base.jpg b/models/Reference/stabilityai--stable-diffusion-2-1.jpg similarity index 100% rename from models/Reference/stabilityai--stable-diffusion-2.1-base.jpg rename to models/Reference/stabilityai--stable-diffusion-2-1.jpg diff --git a/models/Reference/runwayml--stable-diffusion-v1-5.jpg b/models/Reference/v1-5-pruned-fp16-emaonly.jpg similarity index 100% rename from models/Reference/runwayml--stable-diffusion-v1-5.jpg rename to models/Reference/v1-5-pruned-fp16-emaonly.jpg diff --git a/modules/modelloader.py b/modules/modelloader.py index 4084e41f5..ab5f989c6 100644 --- a/modules/modelloader.py +++ b/modules/modelloader.py @@ -280,19 +280,16 @@ def find_diffuser(name: str): def get_reference_opts(name: str): - reference_models = shared.readfile(os.path.join('html', 'reference.json'), silent=False) model_opts = {} - for v in reference_models.values(): - reference_name = v.get('path', '').split('@')[0].split('.')[0] - if reference_name == name: + for k, v in shared.reference_models.items(): + model_name = os.path.splitext(v.get('path', '').split('@')[0])[0] + if k == name or model_name == name: model_opts = v break if not model_opts: shared.log.error(f'Reference: model="{name}" not found') return {} - from modules import styles - styles.reference_style = model_opts.get('extras', None) - shared.log.debug(f'Reference: model="{name}" {styles.reference_style}') + shared.log.debug(f'Reference: model="{name}" {model_opts.get("extras", None)}') return model_opts diff --git a/modules/prompt_parser_diffusers.py b/modules/prompt_parser_diffusers.py index ad3e29b62..3fd60d394 100644 --- a/modules/prompt_parser_diffusers.py +++ b/modules/prompt_parser_diffusers.py @@ -242,11 +242,12 @@ def get_weighted_text_embeddings(pipe, prompt: str = "", neg_prompt: str = "", c .argmax(dim=-1), ] else: - pooled_prompt_embeds = embedding_providers[-1].get_pooled_embeddings(texts=[prompt_2], device=device) if \ - prompt_embeds[-1].shape[-1] > 768 else None - negative_pooled_prompt_embeds = embedding_providers[-1].get_pooled_embeddings(texts=[neg_prompt_2], - device=device) if \ - negative_prompt_embeds[-1].shape[-1] > 768 else None + try: + pooled_prompt_embeds = embedding_providers[-1].get_pooled_embeddings(texts=[prompt_2], device=device) if prompt_embeds[-1].shape[-1] > 768 else None + negative_pooled_prompt_embeds = embedding_providers[-1].get_pooled_embeddings(texts=[neg_prompt_2], device=device) if negative_prompt_embeds[-1].shape[-1] > 768 else None + except Exception: + pooled_prompt_embeds = None + negative_pooled_prompt_embeds = None prompt_embeds = torch.cat(prompt_embeds, dim=-1) if len(prompt_embeds) > 1 else prompt_embeds[0] negative_prompt_embeds = torch.cat(negative_prompt_embeds, dim=-1) if len(negative_prompt_embeds) > 1 else \ diff --git a/modules/sd_models.py b/modules/sd_models.py index 7a5696f6b..af8f86bf7 100644 --- a/modules/sd_models.py +++ b/modules/sd_models.py @@ -884,7 +884,7 @@ def load_diffuser(checkpoint_info=None, already_loaded_state_dict=None, timer=No model_name = modelloader.find_diffuser(ckpt_basename) if model_name is not None: shared.log.info(f'Load model {op}: {model_name}') - model_file = modelloader.download_diffusers_model(hub_id=model_name) + model_file = modelloader.download_diffusers_model(hub_id=model_name, variant=diffusers_load_config.get('variant', None)) try: shared.log.debug(f'Model load {op} config: {diffusers_load_config}') sd_model = diffusers.DiffusionPipeline.from_pretrained(model_file, **diffusers_load_config) @@ -914,15 +914,21 @@ def load_diffuser(checkpoint_info=None, already_loaded_state_dict=None, timer=No if 'variant' not in diffusers_load_config and any('diffusion_pytorch_model.fp16' in f for f in files): # deal with diffusers lack of variant fallback when loading diffusers_load_config['variant'] = 'fp16' if model_type in ['Stable Cascade']: # forced pipeline - # TODO experimental stable cascade - try: - shared.log.debug(f'StableCascade experimental: args={diffusers_load_config} device={devices.device} dtype={devices.dtype}') + try: # this is horrible special-case handling for stable-cascade multi-stage pipeline with variants and non-standard revision diffusers_load_config.pop("vae", None) - diffusers_load_config.pop("variant", None) - decoder = diffusers.StableCascadeDecoderPipeline.from_pretrained("stabilityai/stable-cascade", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/44", **diffusers_load_config) - shared.log.debug(f'StableCascade decoder: scale={decoder.latent_dim_scale}') - prior = diffusers.StableCascadePriorPipeline.from_pretrained("stabilityai/stable-cascade-prior", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/2", **diffusers_load_config) - shared.log.debug(f'StableCascade prior: scale={prior.resolution_multiple}') + diffusers_load_config["variant"] = 'bf16' + if 'lite' in checkpoint_info.name: + decoder_unet = diffusers.models.StableCascadeUNet.from_pretrained("stabilityai/stable-cascade", subfolder="decoder_lite", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/44", **diffusers_load_config) + decoder = diffusers.StableCascadeDecoderPipeline.from_pretrained("stabilityai/stable-cascade", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/44", decoder=decoder_unet, **diffusers_load_config) + shared.log.debug(f'StableCascade lite decoder: scale={decoder.latent_dim_scale}') + prior_unet = diffusers.models.StableCascadeUNet.from_pretrained("stabilityai/stable-cascade-prior", subfolder="prior_lite", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/2", **diffusers_load_config) + prior = diffusers.StableCascadePriorPipeline.from_pretrained("stabilityai/stable-cascade-prior", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/2", prior=prior_unet, **diffusers_load_config) + shared.log.debug(f'StableCascade lite prior: scale={prior.resolution_multiple}') + else: + decoder = diffusers.StableCascadeDecoderPipeline.from_pretrained("stabilityai/stable-cascade", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/44", **diffusers_load_config) + shared.log.debug(f'StableCascade decoder: scale={decoder.latent_dim_scale}') + prior = diffusers.StableCascadePriorPipeline.from_pretrained("stabilityai/stable-cascade-prior", cache_dir=shared.opts.diffusers_dir, revision="refs/pr/2", **diffusers_load_config) + shared.log.debug(f'StableCascade prior: scale={prior.resolution_multiple}') sd_model = diffusers.StableCascadeCombinedPipeline( tokenizer=decoder.tokenizer, text_encoder=decoder.text_encoder, diff --git a/modules/shared.py b/modules/shared.py index f3f6fe904..f55072512 100644 --- a/modules/shared.py +++ b/modules/shared.py @@ -932,6 +932,7 @@ log.info(f'Engine: backend={backend} compute={devices.backend} device={devices.g log.info(f'Device: {print_dict(devices.get_gpu_info())}') prompt_styles = modules.styles.StyleDatabase(opts) +reference_models = readfile(os.path.join('html', 'reference.json')) cmd_opts.disable_extension_access = (cmd_opts.share or cmd_opts.listen or (cmd_opts.server_name or False)) and not cmd_opts.insecure devices.device, devices.device_interrogate, devices.device_gfpgan, devices.device_esrgan, devices.device_codeformer = (devices.cpu if any(y in cmd_opts.use_cpu for y in [x, 'all']) else devices.get_optimal_device() for x in ['sd', 'interrogate', 'gfpgan', 'esrgan', 'codeformer']) devices.onnx = [opts.onnx_execution_provider] diff --git a/modules/styles.py b/modules/styles.py index 1fe97c113..1f0ece4cf 100644 --- a/modules/styles.py +++ b/modules/styles.py @@ -9,7 +9,6 @@ import random from modules import files_cache, shared -reference_style = None class Style(): def __init__(self, name: str, desc: str = "", prompt: str = "", negative_prompt: str = "", extra: str = "", wildcards: str = "", filename: str = "", preview: str = "", mtime: float = 0): @@ -59,14 +58,24 @@ def apply_wildcards_to_prompt(prompt, all_wildcards): return prompt +def get_reference_style(): + name = shared.sd_model.sd_checkpoint_info.name + name = name.replace('\\', '/').replace('Diffusers/', '') + for k, v in shared.reference_models.items(): + model_file = os.path.splitext(v.get('path', '').split('@')[0])[0].replace('huggingface/', '') + if k == name or model_file == name: + return v.get('extras', None) + return None + + def apply_styles_to_extra(p, style: Style): - global reference_style # pylint: disable=global-statement if style is None: return name_map = { 'sampler': 'sampler_name', } from modules.generation_parameters_copypaste import parse_generation_parameters + reference_style = get_reference_style() extra = parse_generation_parameters(reference_style) if shared.opts.extra_network_reference else {} extra.update(parse_generation_parameters(style.extra)) extra.pop('Prompt', None) @@ -87,7 +96,6 @@ def apply_styles_to_extra(p, style: Style): else: skipped.append(f'{k}={v}') shared.log.debug(f'Applying style: name="{style.name}" extra={fields} skipped={skipped} reference={True if reference_style else False}') - # reference_style = None class StyleDatabase: diff --git a/modules/ui_extra_networks.py b/modules/ui_extra_networks.py index ead5800c7..985465ee8 100644 --- a/modules/ui_extra_networks.py +++ b/modules/ui_extra_networks.py @@ -297,14 +297,14 @@ class ExtraNetworksPage: if os.path.join('models', 'Reference') in path: return path exts = ["jpg", "jpeg", "png", "webp", "tiff", "jp2"] + reference_path = os.path.abspath(os.path.join('models', 'Reference')) + files = list(files_cache.list_files(reference_path, ext_filter=exts, recursive=False)) if shared.opts.diffusers_dir in path: path = os.path.relpath(path, shared.opts.diffusers_dir) - reference_path = os.path.abspath(os.path.join('models', 'Reference')) fn = os.path.join(reference_path, path.replace('models--', '').replace('\\', '/').split('/')[0]) - files = list(files_cache.list_files(reference_path, ext_filter=exts, recursive=False)) else: fn = os.path.splitext(path)[0] - files = list(files_cache.list_files(os.path.dirname(path), ext_filter=exts, recursive=False)) + files += list(files_cache.list_files(os.path.dirname(path), ext_filter=exts, recursive=False)) for file in [f'{fn}{mid}{ext}' for ext in exts for mid in ['.thumb.', '.', '.preview.']]: if file in files: if '.thumb.' not in file: @@ -324,6 +324,7 @@ class ExtraNetworksPage: possible_paths = list(set([os.path.dirname(item['filename']) for item in items] + [reference_path])) exts = ["jpg", "jpeg", "png", "webp", "tiff", "jp2"] all_previews = list(files_cache.list_files(*possible_paths, ext_filter=exts, recursive=False)) + all_previews_fn = [os.path.basename(x) for x in all_previews] for item in items: if item.get('preview', None) is not None: continue @@ -336,11 +337,13 @@ class ExtraNetworksPage: model_path = os.path.join(shared.opts.diffusers_dir, match[0]) item['local_preview'] = f'{os.path.join(model_path, match[1])}.{shared.opts.samples_format}' all_previews += list(files_cache.list_files(model_path, ext_filter=exts, recursive=False)) + base = os.path.basename(base) for file in [f'{base}{mid}{ext}' for ext in exts for mid in ['.thumb.', '.', '.preview.']]: - if file in all_previews: + if file in all_previews_fn: + file_idx = all_previews_fn.index(os.path.basename(file)) if '.thumb.' not in file: - self.missing_thumbs.append(file) - item['preview'] = self.link_preview(file) + self.missing_thumbs.append(all_previews[file_idx]) + item['preview'] = self.link_preview(all_previews[file_idx]) break if item.get('preview', None) is None: item['preview'] = self.link_preview('html/card-no-preview.png') diff --git a/modules/ui_extra_networks_checkpoints.py b/modules/ui_extra_networks_checkpoints.py index ab11f0f1d..75ad968ed 100644 --- a/modules/ui_extra_networks_checkpoints.py +++ b/modules/ui_extra_networks_checkpoints.py @@ -15,8 +15,7 @@ class ExtraNetworksPageCheckpoints(ui_extra_networks.ExtraNetworksPage): shared.refresh_checkpoints() def list_reference(self): # pylint: disable=inconsistent-return-statements - reference_models = shared.readfile(os.path.join('html', 'reference.json')) - for k, v in reference_models.items(): + for k, v in shared.reference_models.items(): if shared.backend != shared.Backend.DIFFUSERS: if not v.get('original', False): continue diff --git a/requirements.txt b/requirements.txt index 996f09d40..2b4ad995e 100644 --- a/requirements.txt +++ b/requirements.txt @@ -47,7 +47,7 @@ opencv-contrib-python-headless==4.9.0.80 diffusers==0.26.3 einops==0.4.1 gradio==3.43.2 -huggingface_hub==0.20.3 +huggingface_hub==0.21.4 numexpr==2.8.8 numpy==1.26.4 numba==0.59.0 @@ -55,7 +55,7 @@ pandas protobuf==3.20.3 pytorch_lightning==1.9.4 tokenizers==0.15.2 -transformers==4.38.1 +transformers==4.38.2 tomesd==0.1.3 urllib3==1.26.18 Pillow==10.2.0 diff --git a/wiki b/wiki index 65d62097b..c452d6e07 160000 --- a/wiki +++ b/wiki @@ -1 +1 @@ -Subproject commit 65d62097b8d0ed85b9bd79b1a0ab437708e7c360 +Subproject commit c452d6e07bdaec618161d7d5fa2cf30c41f180bb