diff --git a/CHANGELOG.md b/CHANGELOG.md index 7ade3aaa9..a51426232 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,9 +20,10 @@ Some highlights: [OpenVINO](https://github.com/vladmandic/automatic/wiki/OpenVIN download using built-in **Huggingface** downloader: `segmind/SSD-1B` - new model type: [LCM: Latent Consistency Models](https://github.com/openai/consistency_models) near-instant generate in a as little as 3 steps! - combined with OpenVINO, generate on CPU takes less than 10 seconds: + combined with OpenVINO, generate on CPU takes less than 5-10 seconds: and absolute beast when combined with **HyperTile** and **TAESD** decoder resulting in **28 FPS** (on RTX4090 for batch 16x16 at 512px) + note: set sampler to **Default** before loading model as LCM comes with its own *LCMScheduler* sampler download using built-in **Huggingface** downloader: `SimianLuo/LCM_Dreamshaper_v7` - support for **Custom pipelines**, thanks @disty0 download using built-in **Huggingface** downloader diff --git a/modules/processing.py b/modules/processing.py index 369001598..3a5a4b435 100644 --- a/modules/processing.py +++ b/modules/processing.py @@ -1123,6 +1123,7 @@ class StableDiffusionProcessingTxt2Img(StableDiffusionProcessing): if shared.opts.sd_vae_sliced_encode and len(decoded_samples) > 1: samples = torch.stack([self.sd_model.get_first_stage_encoding(self.sd_model.encode_first_stage(torch.unsqueeze(resized_sample, 0)))[0] for resized_sample in resized_samples]) else: + # TODO add TEASD support samples = self.sd_model.get_first_stage_encoding(self.sd_model.encode_first_stage(resized_samples)) image_conditioning = self.img2img_image_conditioning(resized_samples, samples) else: diff --git a/modules/processing_diffusers.py b/modules/processing_diffusers.py index a3dda0dae..b9b461af9 100644 --- a/modules/processing_diffusers.py +++ b/modules/processing_diffusers.py @@ -14,7 +14,7 @@ import modules.sd_vae as sd_vae import modules.taesd.sd_vae_taesd as sd_vae_taesd import modules.images as images import modules.errors as errors -from modules.processing import StableDiffusionProcessing +from modules.processing import StableDiffusionProcessing, create_random_tensors import modules.prompt_parser_diffusers as prompt_parser_diffusers from modules.sd_hijack_hypertile import hypertile_set @@ -105,7 +105,7 @@ def process_diffusers(p: StableDiffusionProcessing, seeds, prompts, negative_pro devices.torch_gc() if not shared.cmd_opts.lowvram and not shared.opts.diffusers_seq_cpu_offload: model.vae.to(devices.device) - encoded = model.vae.encode(image.to(model.vae.device, model.vae.dtype)) + encoded = model.vae.encode(image.to(model.vae.device, model.vae.dtype)).latent_dist.sample() if shared.opts.diffusers_move_unet and not getattr(model, 'has_accelerate', False): model.unet.to(unet_device) return encoded @@ -147,6 +147,9 @@ def process_diffusers(p: StableDiffusionProcessing, seeds, prompts, negative_pro shared.state.job = prev_job return imgs + def t(x): + return f"\033[34m{str(tuple(x.shape)).ljust(24)}\033[0m (\033[31mmin {x.amin().item():+.4f}\033[0m / \033[32mmean {x.mean().item():+.4f}\033[0m / \033[33mmax {x.amax().item():+.4f}\033[0m)" + def vae_encode(image, model, full_quality=True): # pylint: disable=unused-variable if shared.state.interrupted or shared.state.skipped: return [] @@ -155,6 +158,7 @@ def process_diffusers(p: StableDiffusionProcessing, seeds, prompts, negative_pro return [] tensor = TF.to_tensor(image.convert("RGB")).unsqueeze(0).to(devices.device, devices.dtype_vae) if full_quality: + tensor = tensor * 2 - 1 latents = full_vae_encode(image=tensor, model=shared.sd_model) else: latents = taesd_vae_encode(image=tensor) @@ -198,6 +202,12 @@ def process_diffusers(p: StableDiffusionProcessing, seeds, prompts, negative_pro width = 8 * math.ceil(p.init_images[0].width / 8) height = 8 * math.ceil(p.init_images[0].height / 8) task_args = {"image": p.init_images, "mask_image": p.mask, "strength": p.denoising_strength, "height": height, "width": width} + if model.__class__.__name__ == 'LatentConsistencyModelPipeline' and hasattr(p, 'init_images') and len(p.init_images) > 0: + init_latents = [vae_encode(image, model=shared.sd_model, full_quality=p.full_quality).squeeze(dim=0) for image in p.init_images] + init_latent = torch.stack(init_latents, dim=0).to(shared.device) + init_noise = p.denoising_strength * create_random_tensors(init_latent.shape[1:], seeds=p.all_seeds, subseeds=p.all_subseeds, subseed_strength=p.subseed_strength, p=p) + init_latent = (1 - p.denoising_strength) * init_latent + init_noise + task_args = {"latents": init_latent.to(model.dtype), "width": p.width, "height": p.height } return task_args def set_pipeline_args(model, prompts: list, negative_prompts: list, prompts_2: typing.Optional[list]=None, negative_prompts_2: typing.Optional[list]=None, desc:str='', **kwargs): @@ -268,6 +278,8 @@ def process_diffusers(p: StableDiffusionProcessing, seeds, prompts, negative_pro clean = args.copy() clean.pop('callback', None) clean.pop('callback_steps', None) + if 'latents' in clean: + clean['latents'] = clean['latents'].shape if 'image' in clean: clean['image'] = type(clean['image']) if 'mask_image' in clean: @@ -343,7 +355,8 @@ def process_diffusers(p: StableDiffusionProcessing, seeds, prompts, negative_pro if shared.opts.diffusers_move_base and not getattr(shared.sd_model, 'has_accelerate', False): shared.sd_model.to(devices.device) - is_img2img = bool(sd_models.get_diffusers_task(shared.sd_model) == sd_models.DiffusersTaskType.IMAGE_2_IMAGE or sd_models.get_diffusers_task(shared.sd_model) == sd_models.DiffusersTaskType.INPAINTING) + is_img2img = bool(sd_models.get_diffusers_task(shared.sd_model) == sd_models.DiffusersTaskType.IMAGE_2_IMAGE or + sd_models.get_diffusers_task(shared.sd_model) == sd_models.DiffusersTaskType.INPAINTING) use_refiner_start = bool(is_refiner_enabled and not p.is_hr_pass and not is_img2img and p.refiner_start > 0 and p.refiner_start < 1) use_denoise_start = bool(is_img2img and p.refiner_start > 0 and p.refiner_start < 1) diff --git a/modules/taesd/sd_vae_taesd.py b/modules/taesd/sd_vae_taesd.py index 452f78eb0..82817695e 100644 --- a/modules/taesd/sd_vae_taesd.py +++ b/modules/taesd/sd_vae_taesd.py @@ -80,5 +80,6 @@ def encode(image): taesd_models[f'{model_class}-encoder'] = TAESD(encoder_path=model_path, decoder_path=None) vae = taesd_models[f'{model_class}-encoder'] vae.to(devices.device, devices.dtype_vae) - latents = vae.encoder(image).detach() - return latents + # image = vae.scale_latents(image) + latents = vae.encoder(image) + return latents.detach()