add sdxl-turbo

This commit is contained in:
Vladimir Mandic
2023-11-29 16:32:12 -05:00
parent d8d8f58072
commit f08b4e5c23
10 changed files with 39 additions and 31 deletions
+19 -16
View File
@@ -1,25 +1,28 @@
# Change Log for SD.Next
## Update for 2023-11-28
Note: Release pending `diffusers==0.24`
## Update for 2023-11-29
- **Diffusers**
- [SDXL Turbo](https://huggingface.co/stabilityai/sdxl-turbo) support
- just set CFG scale (0.0-1.0) and steps (1-3) to a very low value
- compatible with original StabilityAI SDXL-Turbo or any of the newer merges
- download safetensors or select from networks -> reference
- **HDR latent control**, based on [article](https://huggingface.co/blog/TimothyAlexisVass/explaining-the-sdxl-latent-space#long-prompts-at-high-guidance-scales-becoming-possible)
- In *Advanced* params
- Allows control of *latent clamping*, *color centering* and *range maximimization*
- Supported by *XYZ grid*
- in *Advanced* params
- allows control of *latent clamping*, *color centering* and *range maximimization*
- supported by *XYZ grid*
- **IP adapter**
- Lightweight implementation of T2I adapters which can guide generation towards specific image style
- Supports most T2I models, not limited to SD 1.5
- Models are auto-downloaded on first use
- For IP adapter support in Original backend, use standard *ControlNet* extension
- lightweight implementation of T2I adapters which can guide generation towards specific image style
- supports most T2I models, not limited to SD 1.5
- models are auto-downloaded on first use
- for IP adapter support in Original backend, use standard *ControlNet* extension
- **AnimateDiff**
- Lightweight implementation of AnimateDiff basic models *(1.4, 1.5, 1.5.2)*
- Supports SD 1.5 only
- Models are auto-downloaded on first use
- For AnimateDiff support in Original backend, use standard *AnimateDiff* extension
- **Kandinsky 3** support
- lightweight implementation of AnimateDiff basic models *(1.4, 1.5, 1.5.2)*
- supports SD 1.5 only
- models are auto-downloaded on first use
- *note*: AnimateDiff can be combined with IP-Adapter for even better results!
- for AnimateDiff support in Original backend, use standard *AnimateDiff* extension
- [Kandinsky 3](https://huggingface.co/kandinsky-community/kandinsky-3) support
- download using built-in model downloader or simply select from networks -> reference
- this model is absolutely massive at 27.5GB at fp16, so be patient
- model params count is at 11.9B (compared to SD-XL at 3.3B) and its trained on mixed resolutions from 256px to 1024px
@@ -27,7 +30,7 @@ Note: Release pending `diffusers==0.24`
- better autodetection of *inpaint* and *instruct* pipelines
- support long seconary prompt for refiner
- **Model merge**
- add sd-xl rebasin support, thanks @AI-Casanova
- add **SD-XL ReBasin** support, thanks @AI-Casanova
- **General**
- further UI optimizations for **mobile devices**, thanks @iDeNoh
- log level defaults to info for console and debug for log file
+3 -2
View File
@@ -167,8 +167,9 @@ def load_networks(names, te_multipliers=None, unet_multipliers=None, dyn_dims=No
if recompile_model:
shared.compiled_model_state.lora_model.append(f"{name}:{te_multipliers[i] if te_multipliers else 1.0}")
if shared.backend == shared.Backend.DIFFUSERS and (os.environ.get('SD_LORA_DIFFUSERS', None)
or getattr(network_on_disk, 'shorthash', None) == 'aaebf6360f7d' # lcm sd15
or getattr(network_on_disk, 'shorthash', None) == '3d18b05e4f56' # lcm sdxl
or getattr(network_on_disk, 'shorthash', '').lower() == 'aaebf6360f7d' # sd15-lcm
or getattr(network_on_disk, 'shorthash', '').lower() == '3d18b05e4f56' # sdxl-lcm
or getattr(network_on_disk, 'shorthash', '').lower() == '813ea5fb1c67' # turbo sdxl-turbo
or (shared.opts.cuda_compile and shared.opts.cuda_compile_backend == "openvino_fx")):
# OpenVINO only works with Diffusers LoRa loading.
net = load_diffusers(name, network_on_disk, lora_scale=te_multipliers[i] if te_multipliers else 1.0)
+5
View File
@@ -14,6 +14,11 @@
"desc": "Stable Diffusion XL (SDXL) is the latest AI image generation model that is tailored towards more photorealistic outputs with more detailed imagery and composition compared to previous SD models, including SD 2.1. It can make realistic faces, legible text within the images, and better image composition, all while using shorter and simpler prompts at a greatly increased base resolution of 1024x1024. Just like its predecessors, SDXL has the ability to generate image variations using image-to-image prompting, inpainting (reimagining of the selected parts of an image), and outpainting (creating new parts that lie outside the image borders).",
"preview": "stabilityai--stable-diffusion-xl-base-1.0.jpg"
},
"StabilityAI SD-XL Turbo": {
"path": "stabilityai/sdxl-turbo",
"desc": "SDXL-Turbo is a distilled version of SDXL 1.0, trained for real-time synthesis. SDXL-Turbo is based on a novel training method called Adversarial Diffusion Distillation (ADD) (see the technical report), which allows sampling large-scale foundational image diffusion models in 1 to 4 steps at high image quality. This approach uses score distillation to leverage large-scale off-the-shelf image diffusion models as a teacher signal and combines this with an adversarial loss to ensure high image fidelity even in the low-step regime of one or two sampling steps.",
"preview": "stabilityai--sdxl-turbo.jpg"
},
"Segmind SSD-1B": {
"path": "segmind/SSD-1B",
"desc": "The Segmind Stable Diffusion Model (SSD-1B) offers a compact, efficient, and distilled version of the SDXL model. At 50% smaller and 60% faster than Stable Diffusion XL (SDXL), it provides quick and seamless performance without sacrificing image quality.",
Binary file not shown.

After

Width:  |  Height:  |  Size: 87 KiB

+1
View File
@@ -192,6 +192,7 @@ class StableDiffusionProcessing:
self.s_tmin = shared.opts.s_tmin
self.s_tmax = float('inf') # not representable as a standard ui option
shared.opts.data['clip_skip'] = clip_skip
self.task_args = {}
# TODO a1111 compatibility items
self.refiner_switch_at = 0
self.hr_prompt = ''
+1
View File
@@ -467,6 +467,7 @@ def process_diffusers(p: StableDiffusionProcessing, seeds, prompts, negative_pro
negative_prompts_2=[p.refiner_negative] if len(p.refiner_negative) > 0 else negative_prompts,
num_inference_steps=calculate_base_steps(),
eta=shared.opts.scheduler_eta,
guidance_scale=p.cfg_scale,
guidance_rescale=p.diffusers_guidance_rescale,
denoising_start=0 if use_refiner_start else p.refiner_start if use_denoise_start else None,
denoising_end=p.refiner_start if use_refiner_start else 1 if use_denoise_start else None,
+2 -3
View File
@@ -332,7 +332,7 @@ def create_sampler_and_steps_selection(choices, tabname):
with FormRow(elem_classes=['flex-break']):
sampler_index = gr.Dropdown(label='Sampling method', elem_id=f"{tabname}_sampling", choices=[x.name for x in choices], value='Default', type="index")
steps = gr.Slider(minimum=0, maximum=99, step=1, label="Sampling steps", elem_id=f"{tabname}_steps", value=20)
steps = gr.Slider(minimum=1, maximum=99, step=1, label="Sampling steps", elem_id=f"{tabname}_steps", value=20)
if modules.shared.backend == modules.shared.Backend.ORIGINAL:
with FormRow(elem_classes=['flex-break']):
choices = ['brownian noise', 'discard penultimate sigma']
@@ -354,7 +354,6 @@ def create_sampler_and_steps_selection(choices, tabname):
values += ['low order'] if opts.data.get('schedulers_use_loworder', True) else []
sampler_options = gr.CheckboxGroup(label='Sampler options', choices=choices, value=values, type='value')
sampler_options.change(fn=set_sampler_diffuser_options, inputs=[sampler_options], outputs=[])
return steps, sampler_index
@@ -419,7 +418,7 @@ def create_ui(startup_timer = None):
with gr.Accordion(open=False, label="Advanced", elem_id="txt2img_advanced", elem_classes=["small-accordion"]):
with gr.Group():
with FormRow():
cfg_scale = gr.Slider(minimum=1.0, maximum=30.0, step=0.1, label='CFG scale', value=6.0, elem_id="txt2img_cfg_scale")
cfg_scale = gr.Slider(minimum=0.0, maximum=30.0, step=0.1, label='CFG scale', value=6.0, elem_id="txt2img_cfg_scale")
clip_skip = gr.Slider(label='CLIP skip', value=1, minimum=1, maximum=14, step=1, elem_id='txt2img_clip_skip', interactive=True)
with FormRow():
image_cfg_scale = gr.Slider(minimum=1.0, maximum=30.0, step=0.1, label='Secondary CFG scale', value=6.0, elem_id="txt2img_image_cfg_scale")
+1 -1
View File
@@ -51,7 +51,7 @@ requests==2.31.0
tqdm==4.66.1
accelerate==0.24.1
opencv-python-headless==4.7.0.72
diffusers==0.23.1
diffusers==0.24.0
einops==0.4.1
gradio==3.43.2
huggingface_hub==0.19.4
+6 -8
View File
@@ -52,9 +52,6 @@ def set_adapter(name: str = None):
if shared.backend != shared.Backend.DIFFUSERS:
shared.log.warning('AnimateDiff: not in diffusers mode')
return
if shared.sd_model_type != 'sd':
shared.log.warning(f'AnimateDiff: unsupported model type: {shared.sd_model.__class__.__name__}')
return
global motion_adapter, loaded_adapter, orig_pipe # pylint: disable=global-statement
adapter_name = name if name is not None and isinstance(name, str) else loaded_adapter
if adapter_name is None or adapter_name == 'None' or shared.sd_model is None:
@@ -65,6 +62,9 @@ def set_adapter(name: str = None):
shared.sd_model = orig_pipe
orig_pipe = None
return
if shared.sd_model_type != 'sd':
shared.log.warning(f'AnimateDiff: unsupported model type: {shared.sd_model.__class__.__name__}')
return
if motion_adapter is not None and loaded_adapter == adapter_name:
shared.log.info(f'AnimateDiff cache: adapter="{adapter_name}"')
return
@@ -139,11 +139,9 @@ class Script(scripts.Script):
shared.sd_model.set_adapters([lora], adapter_weights=[strength])
p.extra_generation_params['AnimateDiff Lora'] = f'{lora}:{strength}'
p.do_not_save_grid = True
p.task_args = {
'num_frames': frames,
'output_type': 'np', # TODO: AnimateDiff use latents and update vae_decode
'num_inference_steps': p.steps,
}
p.task_args['num_frames'] = frames
p.task_args['output_type'] = 'np' # TODO: AnimateDiff use latents and update vae_decode
p.task_args['num_inference_steps'] = p.steps
def postprocess(self, p: processing.StableDiffusionProcessing, processed: processing.Processed, adapter_index, frames, lora_index, strength, override, create_gif, duration, loop): # pylint: disable=arguments-differ, unused-argument
if not create_gif or len(processed.images) < 2:
+1 -1
View File
@@ -106,5 +106,5 @@ class Script(scripts.Script):
else:
shared.log.debug(f'IP adapter cache: adapter="{model}" scale={scale} image={image}')
shared.sd_model.set_ip_adapter_scale(scale)
p.task_args = { 'ip_adapter_image': p.batch_size * [image] }
p.task_args['ip_adapter_image'] = p.batch_size * [image]
p.extra_generation_params["IP Adapter"] = f'{adapter}:{scale}'