diff --git a/CHANGELOG.md b/CHANGELOG.md index 8988d9b12..dd7a3f26c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -179,6 +179,7 @@ Further details: - fix *differenital diffusion* for manual mask, thanks @23pennies - fix ipadapter apply/unapply on batch runs - fix control with multiple units and override images + - fix control with hires - fix control-lllite - fix font fallback, thanks @NetroScript - update civitai downloader to handler new metadata diff --git a/modules/processing.py b/modules/processing.py index 04cc81d7d..185077483 100644 --- a/modules/processing.py +++ b/modules/processing.py @@ -324,9 +324,6 @@ def process_images_inner(p: StableDiffusionProcessing) -> Processed: def infotext(index): # pylint: disable=function-redefined # noqa: F811 return create_infotext(p, p.prompts, p.seeds, p.subseeds, index=index, all_negative_prompts=p.negative_prompts) - if hasattr(shared.sd_model, 'restore_pipeline') and shared.sd_model.restore_pipeline is not None: - shared.sd_model.restore_pipeline() - for i, x_sample in enumerate(x_samples_ddim): if hasattr(p, 'recursion'): continue @@ -384,6 +381,9 @@ def process_images_inner(p: StableDiffusionProcessing) -> Processed: del x_samples_ddim devices.torch_gc() + if hasattr(shared.sd_model, 'restore_pipeline') and shared.sd_model.restore_pipeline is not None: + shared.sd_model.restore_pipeline() + t1 = time.time() shared.log.info(f'Processed: images={len(output_images)} time={t1 - t0:.2f} its={(p.steps * len(output_images)) / (t1 - t0):.2f} memory={memstats.memory_stats()}') diff --git a/modules/processing_diffusers.py b/modules/processing_diffusers.py index 14cd83a43..29da4a5ab 100644 --- a/modules/processing_diffusers.py +++ b/modules/processing_diffusers.py @@ -298,7 +298,7 @@ def process_diffusers(p: processing.StableDiffusionProcessing): clean['generator'] = generator_device clean['parser'] = parser for k, v in clean.items(): - if isinstance(v, torch.Tensor): + if isinstance(v, torch.Tensor) or isinstance(v, np.ndarray) or (isinstance(v, list) and len(v) > 0 and (isinstance(v[0], torch.Tensor) or isinstance(v[0], np.ndarray))): clean[k] = v.shape shared.log.debug(f'Diffuser pipeline: {model.__class__.__name__} task={sd_models.get_diffusers_task(model)} set={clean}') if p.hdr_clamp or p.hdr_maximize or p.hdr_brightness != 0 or p.hdr_color != 0 or p.hdr_sharpen != 0: @@ -453,8 +453,6 @@ def process_diffusers(p: processing.StableDiffusionProcessing): if hasattr(shared.sd_model, 'embedding_db') and len(shared.sd_model.embedding_db.embeddings_used) > 0: # register used embeddings p.extra_generation_params['Embeddings'] = ', '.join(shared.sd_model.embedding_db.embeddings_used) - if hasattr(p, 'task_args') and p.task_args.get('image', None) is not None and output is not None: # replace input with output so it can be used by hires/refine - p.task_args['image'] = output.images shared.state.nextjob() if shared.state.interrupted or shared.state.skipped: @@ -479,8 +477,6 @@ def process_diffusers(p: processing.StableDiffusionProcessing): save_intermediate(latents=output.images, suffix="-before-hires") shared.state.job = 'upscale' output.images = resize_hires(p, latents=output.images) - if hasattr(p, 'task_args') and p.task_args.get('image', None) is not None and output is not None: # replace input with output so it can be used by hires/refine - p.task_args['image'] = output.images sd_hijack_hypertile.hypertile_set(p, hr=True) latent_upscale = shared.latent_upscale_modes.get(p.hr_upscaler, None) @@ -497,6 +493,10 @@ def process_diffusers(p: processing.StableDiffusionProcessing): shared.sd_model = sd_models.set_diffuser_pipe(shared.sd_model, sd_models.DiffusersTaskType.IMAGE_2_IMAGE) update_sampler(shared.sd_model, second_pass=True) shared.log.info(f'HiRes: class={shared.sd_model.__class__.__name__} sampler="{p.hr_sampler_name}"') + if p.is_control and hasattr(p, 'task_args') and p.task_args.get('image', None) is not None: + if hasattr(shared.sd_model, "vae") and output.images is not None and len(output.images) > 0: + output.images = processing_vae.vae_decode(latents=output.images, model=shared.sd_model, full_quality=p.full_quality, output_type='pil') # controlnet cannnot deal with latent input + p.task_args['image'] = output.images # replace so hires uses new output sd_models.move_model(shared.sd_model, devices.device) orig_denoise = p.denoising_strength p.denoising_strength = getattr(p, 'hr_denoising_strength', p.denoising_strength) @@ -528,7 +528,6 @@ def process_diffusers(p: processing.StableDiffusionProcessing): except AssertionError as e: shared.log.info(e) p.denoising_strength = orig_denoise - # p.init_images = [] shared.state.job = prev_job shared.state.nextjob() p.is_hr_pass = False @@ -562,6 +561,8 @@ def process_diffusers(p: processing.StableDiffusionProcessing): image = processing_vae.vae_decode(latents=image, model=shared.sd_model, full_quality=p.full_quality, output_type='pil') p.extra_generation_params['Noise level'] = noise_level output_type = 'np' + if hasattr(p, 'task_args') and p.task_args.get('image', None) is not None and output is not None: # replace input with output so it can be used by hires/refine + p.task_args['image'] = image shared.log.info(f'Refiner: class={shared.sd_refiner.__class__.__name__}') refiner_args = set_pipeline_args( model=shared.sd_refiner, diff --git a/modules/processing_vae.py b/modules/processing_vae.py index 793f2aeb8..1120b4ba1 100644 --- a/modules/processing_vae.py +++ b/modules/processing_vae.py @@ -120,10 +120,10 @@ def vae_decode(latents, model, output_type='np', full_quality=True): if not hasattr(model, 'vae'): shared.log.error('VAE not found in model') return [] - if latents.shape[0] == 4 and latents.shape[1] != 4: # likely animatediff latent - latents = latents.permute(1, 0, 2, 3) if len(latents.shape) == 3: # lost a batch dim in hires latents = latents.unsqueeze(0) + if latents.shape[0] == 4 and latents.shape[1] != 4: # likely animatediff latent + latents = latents.permute(1, 0, 2, 3) if full_quality: decoded = full_vae_decode(latents=latents, model=shared.sd_model) else: