From 4535a99fff3d9b95e10e0fa9efe6b14078f89264 Mon Sep 17 00:00:00 2001 From: Disty0 Date: Thu, 3 Aug 2023 17:25:15 +0300 Subject: [PATCH] Model compile support for IPEX --- modules/sd_hijack.py | 38 +++++++++++++++++++++----------------- modules/sd_models.py | 12 ++++++++---- modules/shared.py | 5 +++-- 3 files changed, 32 insertions(+), 23 deletions(-) diff --git a/modules/sd_hijack.py b/modules/sd_hijack.py index 362285f4a..67826cb91 100644 --- a/modules/sd_hijack.py +++ b/modules/sd_hijack.py @@ -174,27 +174,31 @@ class StableDiffusionModelHijack: if m.cond_stage_key == "edit": sd_hijack_unet.hijack_ddpm_edit() + if opts.ipex_optimize and shared.backend == shared.Backend.ORIGINAL: + try: + import intel_extension_for_pytorch as ipex # pylint: disable=import-error, unused-import + m.model.training = False + m.model = ipex.optimize(m.model, dtype=devices.dtype_unet, inplace=True, weights_prepack=False) # pylint: disable=attribute-defined-outside-init + shared.log.info("Applied IPEX Optimize.") + except Exception as err: + shared.log.warning(f"IPEX Optimize not supported: {err}") + if opts.cuda_compile and opts.cuda_compile_mode != 'none' and shared.backend == shared.Backend.ORIGINAL: try: import logging shared.log.info(f"Compiling pipeline={m.model.__class__.__name__} mode={opts.cuda_compile_mode}") - if opts.cuda_compile_mode == 'ipex': - import intel_extension_for_pytorch as ipex # pylint: disable=import-error, unused-import - m.model.training = False - m.model = ipex.optimize(m.model, dtype=devices.dtype_unet, inplace=True, weights_prepack=False) # pylint: disable=attribute-defined-outside-init - else: - import torch._dynamo # pylint: disable=unused-import,redefined-outer-name - log_level = logging.WARNING if opts.cuda_compile_verbose else logging.CRITICAL # pylint: disable=protected-access - if hasattr(torch, '_logging'): - torch._logging.set_logs(dynamo=log_level, aot=log_level, inductor=log_level) # pylint: disable=protected-access - torch._dynamo.config.verbose = opts.cuda_compile_verbose # pylint: disable=protected-access - torch._dynamo.config.suppress_errors = opts.cuda_compile_errors # pylint: disable=protected-access - torch.backends.cudnn.benchmark = True - if opts.cuda_compile_mode == 'hidet': - import hidet - hidet.torch.dynamo_config.use_tensor_core(True) - hidet.torch.dynamo_config.search_space(2) - m.model = torch.compile(m.model, mode="default", backend=opts.cuda_compile_mode, fullgraph=opts.cuda_compile_fullgraph, dynamic=False) + import torch._dynamo # pylint: disable=unused-import,redefined-outer-name + log_level = logging.WARNING if opts.cuda_compile_verbose else logging.CRITICAL # pylint: disable=protected-access + if hasattr(torch, '_logging'): + torch._logging.set_logs(dynamo=log_level, aot=log_level, inductor=log_level) # pylint: disable=protected-access + torch._dynamo.config.verbose = opts.cuda_compile_verbose # pylint: disable=protected-access + torch._dynamo.config.suppress_errors = opts.cuda_compile_errors # pylint: disable=protected-access + torch.backends.cudnn.benchmark = True + if opts.cuda_compile_mode == 'hidet': + import hidet + hidet.torch.dynamo_config.use_tensor_core(True) + hidet.torch.dynamo_config.search_space(2) + m.model = torch.compile(m.model, mode="default", backend=opts.cuda_compile_mode, fullgraph=opts.cuda_compile_fullgraph, dynamic=False) shared.log.info("Model complilation done.") except Exception as err: shared.log.warning(f"Model compile not supported: {err}") diff --git a/modules/sd_models.py b/modules/sd_models.py index fa7002ec0..b9a64b870 100644 --- a/modules/sd_models.py +++ b/modules/sd_models.py @@ -708,7 +708,7 @@ def load_diffuser(checkpoint_info=None, already_loaded_state_dict=None, timer=No sd_model.unet.to(memory_format=torch.channels_last) base_sent_to_cpu=False - if shared.opts.cuda_compile and torch.cuda.is_available(): + if (shared.opts.cuda_compile or shared.opts.ipex_optimize) and torch.cuda.is_available(): if op == 'refiner' and not sd_model.has_accelerate: gpu_vram = memory_stats().get('gpu', {}) free_vram = gpu_vram.get('total', 0) - gpu_vram.get('used', 0) @@ -731,11 +731,15 @@ def load_diffuser(checkpoint_info=None, already_loaded_state_dict=None, timer=No elif not sd_model.has_accelerate: sd_model.to(devices.device) try: - shared.log.info(f"Compiling pipeline={sd_model.__class__.__name__} shape={8 * sd_model.unet.config.sample_size} mode={shared.opts.cuda_compile_mode}") - if shared.opts.cuda_compile_mode == 'ipex': + if shared.opts.ipex_optimize: sd_model.unet.training = False sd_model.unet = torch.xpu.optimize(sd_model.unet, dtype=devices.dtype_unet, inplace=True, weights_prepack=False) # pylint: disable=attribute-defined-outside-init - else: + shared.log.info("Applied IPEX Optimize.") + except Exception as err: + shared.log.warning(f"IPEX Optimize not supported: {err}") + try: + if shared.opts.cuda_compile: + shared.log.info(f"Compiling pipeline={sd_model.__class__.__name__} shape={8 * sd_model.unet.config.sample_size} mode={shared.opts.cuda_compile_mode}") import torch._dynamo # pylint: disable=unused-import,redefined-outer-name log_level = logging.WARNING if shared.opts.cuda_compile_verbose else logging.CRITICAL # pylint: disable=protected-access if hasattr(torch, '_logging'): diff --git a/modules/shared.py b/modules/shared.py index 63f5549b4..fa9aab298 100644 --- a/modules/shared.py +++ b/modules/shared.py @@ -384,12 +384,13 @@ options_templates.update(options_section(('cuda', "Compute Settings"), { "cudnn_benchmark": OptionInfo(False, "Enable full-depth cuDNN benchmark feature"), "cuda_allow_tf32": OptionInfo(True, "Allow TF32 math ops"), "cuda_allow_tf16_reduced": OptionInfo(True, "Allow TF16 reduced precision math ops"), - "cuda_compile": OptionInfo(True if devices.backend == "ipex" else False, "Enable model compile (experimental)"), - "cuda_compile_mode": OptionInfo("ipex" if devices.backend == "ipex" else "none", "Model compile mode (experimental)", gr.Radio, lambda: {"choices": ['none', 'inductor', 'reduce-overhead', 'cudagraphs', 'aot_ts_nvfuser', 'hidet', 'ipex']}), + "cuda_compile": OptionInfo(False, "Enable model compile (experimental)"), + "cuda_compile_mode": OptionInfo("none", "Model compile mode (experimental)", gr.Radio, lambda: {"choices": ['none', 'default', 'inductor', 'reduce-overhead', 'cudagraphs', 'aot_ts_nvfuser', 'hidet', 'max-autotune', 'ipex']}), "cuda_compile_fullgraph": OptionInfo(False, "Model compile fullgraph"), "cuda_compile_verbose": OptionInfo(False, "Model compile verbose mode"), "cuda_compile_errors": OptionInfo(True, "Model compile suppress errors"), "disable_gc": OptionInfo(True, "Disable Torch memory garbage collection"), + "ipex_optimize": OptionInfo(True if devices.backend == "ipex" else False, "Enable IPEX Optimize for Intel GPUs"), "directml_memory_provider": OptionInfo(default_memory_provider, '[DirectML] Memory stats provider', gr.Dropdown, lambda: {"choices": memory_providers}), }))