From a8b2d0b8c97ec75f7f53062cade323856573b45d Mon Sep 17 00:00:00 2001 From: Disty0 Date: Sun, 21 Apr 2024 00:53:31 +0300 Subject: [PATCH] Update defaults and add autodetect for VRAM optimizations --- installer.py | 6 ++++-- modules/sd_models.py | 4 ++-- modules/shared.py | 32 +++++++++++++++++++++++++------- 3 files changed, 31 insertions(+), 11 deletions(-) diff --git a/installer.py b/installer.py index 6b62bd7b3..3d3fc5e30 100644 --- a/installer.py +++ b/installer.py @@ -545,8 +545,10 @@ def check_torch(): elif allow_ipex and (args.use_ipex or shutil.which('sycl-ls') is not None or shutil.which('sycl-ls.exe') is not None or os.environ.get('ONEAPI_ROOT') is not None or os.path.exists('/opt/intel/oneapi') or os.path.exists("C:/Program Files (x86)/Intel/oneAPI") or os.path.exists("C:/oneAPI")): args.use_ipex = True # pylint: disable=attribute-defined-outside-init log.info('Intel OneAPI Toolkit detected') - os.environ.setdefault('NEOReadDebugKeys', '1') - os.environ.setdefault('ClDeviceGlobalMemSizeAvailablePercent', '100') + if os.environ.get("NEOReadDebugKeys", None) is None: + os.environ.setdefault('NEOReadDebugKeys', '1') + if os.environ.get("ClDeviceGlobalMemSizeAvailablePercent", None) is None: + os.environ.setdefault('ClDeviceGlobalMemSizeAvailablePercent', '100') if "linux" in sys.platform: torch_command = os.environ.get('TORCH_COMMAND', 'torch==2.1.0.post0 torchvision==0.16.0.post0 intel-extension-for-pytorch==2.1.20+xpu --extra-index-url https://pytorch-extension.intel.com/release-whl/stable/xpu/us/') os.environ.setdefault('TENSORFLOW_PACKAGE', 'tensorflow==2.15.0 intel-extension-for-tensorflow[xpu]==2.15.0.0') diff --git a/modules/sd_models.py b/modules/sd_models.py index 23053cc51..267f8c407 100644 --- a/modules/sd_models.py +++ b/modules/sd_models.py @@ -709,13 +709,13 @@ def set_diffuser_options(sd_model, vae = None, op: str = 'model'): sd_model.enable_sequential_cpu_offload() sd_model.has_accelerate = True if hasattr(sd_model, "enable_vae_slicing"): - if shared.cmd_opts.lowvram or shared.opts.diffusers_vae_slicing: + if shared.opts.diffusers_vae_slicing: shared.log.debug(f'Setting {op}: enable VAE slicing') sd_model.enable_vae_slicing() else: sd_model.disable_vae_slicing() if hasattr(sd_model, "enable_vae_tiling"): - if shared.cmd_opts.lowvram or shared.opts.diffusers_vae_tiling: + if shared.opts.diffusers_vae_tiling: shared.log.debug(f'Setting {op}: enable VAE tiling') sd_model.enable_vae_tiling() else: diff --git a/modules/shared.py b/modules/shared.py index dfd94e5f0..70fdc764b 100644 --- a/modules/shared.py +++ b/modules/shared.py @@ -19,6 +19,7 @@ from modules.paths import models_path, script_path, data_path, sd_configs_path, from modules.dml import memory_providers, default_memory_provider, directml_do_hijack from modules.onnx_impl import initialize_onnx, execution_providers from modules.zluda import initialize_zluda +from modules.memstats import memory_stats import modules.interrogate import modules.memmon import modules.styles @@ -348,14 +349,31 @@ def temp_disable_extensions(): return disabled -if devices.backend == "cpu": +if not (cmd_opts.lowvram or cmd_opts.medvram): + mem_stat = memory_stats() + if "gpu" in mem_stat: + if mem_stat['gpu']['total'] <= 4: + cmd_opts.lowvram = True + log.info(f"VRAM: Detected={mem_stat['gpu']['total']} GB Optimization=lowvram") + elif mem_stat['gpu']['total'] <= 8: + cmd_opts.medvram = True + log.info(f"VRAM: Detected={mem_stat['gpu']['total']} GB Optimization=medvram") + else: + log.info(f"VRAM: Detected={mem_stat['gpu']['total']} GB Optimization=none") + + +if devices.backend == "directml": # Force BMM for DirectML instead of SDP + cross_attention_optimization_default = "Dynamic Attention BMM" if backend == Backend.DIFFUSERS else "Sub-quadratic" +elif backend == Backend.DIFFUSERS and (cmd_opts.lowvram or cmd_opts.medvram): + cross_attention_optimization_default = "Dynamic Attention SDP" +elif devices.backend == "cpu": cross_attention_optimization_default = "Scaled-Dot-Product" if backend == Backend.DIFFUSERS else "Doggettx's" elif devices.backend == "mps": cross_attention_optimization_default = "Scaled-Dot-Product" if backend == Backend.DIFFUSERS else "Doggettx's" -elif devices.backend == "directml": - cross_attention_optimization_default = "Dynamic Attention BMM" if backend == Backend.DIFFUSERS else "Sub-quadratic" else: # cuda, rocm, ipex cross_attention_optimization_default ="Scaled-Dot-Product" + + if devices.backend == "rocm": sdp_options_default = ['Memory attention', 'Math attention'] #elif devices.backend == "zluda": @@ -485,13 +503,13 @@ options_templates.update(options_section(('diffusers', "Diffusers Settings"), { "diffusers_move_base": OptionInfo(False, "Move base model to CPU when using refiner"), "diffusers_move_unet": OptionInfo(False, "Move base model to CPU when using VAE"), "diffusers_move_refiner": OptionInfo(False, "Move refiner model to CPU when not in use"), - "diffusers_extract_ema": OptionInfo(True, "Use model EMA weights when possible"), + "diffusers_extract_ema": OptionInfo(False, "Use model EMA weights when possible"), "diffusers_generator_device": OptionInfo("GPU", "Generator device", gr.Radio, {"choices": ["GPU", "CPU", "Unset"]}), - "diffusers_model_cpu_offload": OptionInfo(False, "Model CPU offload (--medvram)"), - "diffusers_seq_cpu_offload": OptionInfo(False, "Sequential CPU offload (--lowvram)"), + "diffusers_model_cpu_offload": OptionInfo(cmd_opts.medvram, "Model CPU offload (--medvram)"), + "diffusers_seq_cpu_offload": OptionInfo(cmd_opts.lowvram, "Sequential CPU offload (--lowvram)"), "diffusers_vae_upcast": OptionInfo("default", "VAE upcasting", gr.Radio, {"choices": ['default', 'true', 'false']}), "diffusers_vae_slicing": OptionInfo(True, "VAE slicing"), - "diffusers_vae_tiling": OptionInfo(False, "VAE tiling"), + "diffusers_vae_tiling": OptionInfo(cmd_opts.lowvram or cmd_opts.medvram, "VAE tiling"), "diffusers_model_load_variant": OptionInfo("default", "Preferred Model variant", gr.Radio, {"choices": ['default', 'fp32', 'fp16']}), "diffusers_vae_load_variant": OptionInfo("default", "Preferred VAE variant", gr.Radio, {"choices": ['default', 'fp32', 'fp16']}), "custom_diffusers_pipeline": OptionInfo('', 'Load custom Diffusers pipeline'),