diff --git a/installer.py b/installer.py index 05bbeb5bb..c04fecead 100644 --- a/installer.py +++ b/installer.py @@ -375,10 +375,10 @@ def check_torch(): os.environ.setdefault('NEOReadDebugKeys', '1') os.environ.setdefault('ClDeviceGlobalMemSizeAvailablePercent', '100') if "linux" in sys.platform: - torch_command = os.environ.get('TORCH_COMMAND', 'torch==2.0.1a0 torchvision==0.15.2a0 intel_extension_for_pytorch==2.0.110+xpu -f https://developer.intel.com/ipex-whl-stable-xpu') + torch_command = os.environ.get('TORCH_COMMAND', 'torch==2.0.1a0 torchvision==0.15.2a0 intel_extension_for_pytorch==2.0.110+xpu openvino==2023.1.0.dev20230728 -f https://developer.intel.com/ipex-whl-stable-xpu') os.environ.setdefault('TENSORFLOW_PACKAGE', 'tensorflow==2.13.0 intel-extension-for-tensorflow[gpu]') else: - torch_command = os.environ.get('TORCH_COMMAND', 'torch==2.0.0a0 torchvision==0.15.2a0 intel_extension_for_pytorch==2.0.110+gitba7f6c1 -f https://developer.intel.com/ipex-whl-stable-xpu') + torch_command = os.environ.get('TORCH_COMMAND', 'torch==2.0.0a0 torchvision==0.15.1 intel_extension_for_pytorch==2.0.110+gitba7f6c1 openvino==2023.1.0.dev20230728 -f https://developer.intel.com/ipex-whl-stable-xpu') else: machine = platform.machine() if sys.platform == 'darwin': diff --git a/modules/ipex_specific/__init__.py b/modules/ipex_specific/__init__.py index ecb72641f..f8fdca2eb 100644 --- a/modules/ipex_specific/__init__.py +++ b/modules/ipex_specific/__init__.py @@ -86,3 +86,7 @@ def ipex_init(): ipex_hijacks() ipex_diffusers() + try: + from .openvino import openvino_fx + except Exception: + pass diff --git a/modules/ipex_specific/hijacks.py b/modules/ipex_specific/hijacks.py index 32e075204..05419f426 100644 --- a/modules/ipex_specific/hijacks.py +++ b/modules/ipex_specific/hijacks.py @@ -8,6 +8,40 @@ def ipex_no_cuda(orig_func, *args, **kwargs): # pylint: disable=redefined-outer- orig_func(*args, **kwargs) torch.cuda.is_available = torch.xpu.is_available +#FP32: +original_linear_forward = torch.nn.modules.Linear.forward +def linear_forward(self, input): + if input.dtype != self.weight.data.dtype: + return original_linear_forward(self, input.to(self.weight.data.dtype)) + else: + return original_linear_forward(self, input) + +#Embedding BF16 +original_torch_cat = torch.cat +def torch_cat(input, *args, **kwargs): + if len(input) == 3 and (input[0].dtype != input[1].dtype or input[2].dtype != input[1].dtype): + return original_torch_cat([input[0].to(input[1].dtype), input[1], input[2].to(input[1].dtype)], *args, **kwargs) + else: + return original_torch_cat(input, *args, **kwargs) + +original_conv2d = torch.nn.functional.conv2d +#Diffusers BF16: +def conv2d(input, weight, *args, **kwargs): + if input.dtype != weight.data.dtype: + return original_conv2d(input.to(weight.data.dtype), weight, *args, **kwargs) + else: + return original_conv2d(input, weight, *args, **kwargs) + +original_interpolate = torch.nn.functional.interpolate +#Latent antialias: +def interpolate(input, size=None, scale_factor=None, mode='nearest', align_corners=None, recompute_scale_factor=None, antialias=False): + if antialias: + return original_interpolate(input.to("cpu"), size=size, scale_factor=scale_factor, mode=mode, + align_corners=align_corners, recompute_scale_factor=recompute_scale_factor, antialias=antialias).to(shared.device) + else: + return original_interpolate(input, size=size, scale_factor=scale_factor, mode=mode, + align_corners=align_corners, recompute_scale_factor=recompute_scale_factor, antialias=antialias) + def ipex_hijacks(): #Libraries that blindly uses cuda: #Adetailer: @@ -36,10 +70,6 @@ def ipex_hijacks(): CondFunc('torch.nn.modules.GroupNorm.forward', lambda orig_func, self, input: orig_func(self, input.to(self.weight.data.dtype)), lambda orig_func, self, input: input.dtype != self.weight.data.dtype) - #FP32: - CondFunc('torch.nn.modules.Linear.forward', - lambda orig_func, self, input: orig_func(self, input.to(self.weight.data.dtype)), - lambda orig_func, self, input: input.dtype != self.weight.data.dtype) #Embedding FP32: CondFunc('torch.bmm', lambda orig_func, input, mat2, *args, **kwargs: orig_func(input, mat2.to(input.dtype), *args, **kwargs), @@ -50,14 +80,6 @@ def ipex_hijacks(): orig_func(input.to(weight.data.dtype), normalized_shape, weight, *args, **kwargs), lambda orig_func, input, normalized_shape=None, weight=None, *args, **kwargs: input.dtype != weight.data.dtype and weight is not None) - #Embedding BF16 - CondFunc('torch.cat', - lambda orig_func, input, *args, **kwargs: orig_func([input[0].to(input[1].dtype), input[1], input[2].to(input[1].dtype)], *args, **kwargs), - lambda orig_func, input, *args, **kwargs: len(input) == 3 and (input[0].dtype != input[1].dtype or input[2].dtype != input[1].dtype)) - #Diffusers BF16: - CondFunc('torch.nn.functional.conv2d', - lambda orig_func, input, weight, *args, **kwargs: orig_func(input.to(weight.data.dtype), weight, *args, **kwargs), - lambda orig_func, input, weight, *args, **kwargs: input.dtype != weight.data.dtype) #Functions that does not work with the XPU: #UniPC: @@ -68,10 +90,6 @@ def ipex_hijacks(): CondFunc('torch.Generator', lambda orig_func, device: torch.xpu.Generator(device), lambda orig_func, device: device != torch.device("cpu") and device != "cpu") - #Latent antialias: - CondFunc('torch.nn.functional.interpolate', - lambda orig_func, input, *args, **kwargs: orig_func(input.to("cpu"), *args, **kwargs).to(shared.device), - lambda orig_func, input, size=None, scale_factor=None, mode='nearest', align_corners=None, recompute_scale_factor=None, antialias=False: antialias) #Diffusers Float64 (ARC GPUs doesn't support double or Float64): if not torch.xpu.has_fp64_dtype(): CondFunc('torch.from_numpy', @@ -89,3 +107,9 @@ def ipex_hijacks(): weight if weight is not None else torch.ones(input.size()[1], device=shared.device), bias if bias is not None else torch.zeros(input.size()[1], device=shared.device), *args, **kwargs), lambda orig_func, input, *args, **kwargs: input.device != torch.device("cpu")) + + #Functions that make compile mad with CondFunc: + torch.nn.modules.Linear.forward = linear_forward + torch.cat = torch_cat + torch.nn.functional.conv2d = conv2d + torch.nn.functional.interpolate = interpolate diff --git a/modules/ipex_specific/openvino.py b/modules/ipex_specific/openvino.py new file mode 100644 index 000000000..5101c7d8e --- /dev/null +++ b/modules/ipex_specific/openvino.py @@ -0,0 +1,33 @@ +import os +import torch +import intel_extension_for_pytorch as ipex +from openvino.frontend.pytorch.torchdynamo.execute import execute +from openvino.frontend.pytorch.torchdynamo.partition import Partitioner +from torch._dynamo.backends.common import fake_tensor_unsupported +from torch._dynamo.backends.registry import register_backend +from torch._inductor.compile_fx import compile_fx +from torch.fx.experimental.proxy_tensor import make_fx + +class ModelState: + def __init__(self): + self.recompile = 1 + self.partition_id = 0 + +model_state = ModelState() + +@register_backend +@fake_tensor_unsupported +def openvino_fx(subgraph, example_inputs): + if os.getenv("OPENVINO_TORCH_BACKEND_DEVICE") is None: + os.environ.setdefault("OPENVINO_TORCH_BACKEND_DEVICE", "GPU") + + model = make_fx(subgraph)(*example_inputs) + with torch.no_grad(): + model.eval() + partitioner = Partitioner() + compiled_model = partitioner.make_partitions(model) + + def _call(*args): + res = execute(compiled_model, *args, executor="openvino") + return res + return _call diff --git a/modules/shared.py b/modules/shared.py index 945f2a27f..de66b56c8 100644 --- a/modules/shared.py +++ b/modules/shared.py @@ -384,7 +384,7 @@ options_templates.update(options_section(('cuda', "Compute Settings"), { # "cuda_allow_tf32": OptionInfo(True, "Allow TF32 math ops"), # "cuda_allow_tf16_reduced": OptionInfo(True, "Allow TF16 reduced precision math ops"), "cuda_compile": OptionInfo(False, "Enable model compile (experimental)"), - "cuda_compile_backend": OptionInfo("none", "Model compile backend (experimental)", gr.Radio, lambda: {"choices": ['none', 'inductor', 'cudagraphs', 'aot_ts_nvfuser', 'hidet', 'ipex']}), + "cuda_compile_backend": OptionInfo("none", "Model compile backend (experimental)", gr.Radio, lambda: {"choices": ['none', 'inductor', 'cudagraphs', 'aot_ts_nvfuser', 'hidet', 'ipex', 'openvino_fx']}), "cuda_compile_mode": OptionInfo("default", "Model compile mode (experimental)", gr.Radio, lambda: {"choices": ['default', 'reduce-overhead', 'max-autotune']}), "cuda_compile_fullgraph": OptionInfo(False, "Model compile fullgraph"), "cuda_compile_verbose": OptionInfo(False, "Model compile verbose mode"), diff --git a/webui.sh b/webui.sh index 38fc189b1..59f235aac 100755 --- a/webui.sh +++ b/webui.sh @@ -96,10 +96,6 @@ if [[ ! -z "${ACCELERATE}" ]] && [ ${ACCELERATE}="True" ] && [ -x "$(command -v then echo "Launching accelerate launch.py..." exec accelerate launch --num_cpu_threads_per_process=6 launch.py "$@" -elif [[ "$@" == *"--use-ipex"* ]] && [[ -z "${first_launch}" ]] && [ -x "$(command -v ipexrun)" ] && [ -x "$(command -v sycl-ls)" ] -then - echo "Launching ipexrun launch.py..." - exec ipexrun --multi-task-manager 'taskset' launch.py "$@" else echo "Launching launch.py..." exec "${python_cmd}" launch.py "$@"