fix(rocm): add option to load models without mmap

On ROCm, host-to-device DMA from mmap'd safetensors pages stalls ~1s per
copy, so weights move to the GPU at ~27 MB/s instead of ~28 GB/s. With
offload enabled this re-copies weights every forward, so generation
appears to hang.

Adds `diffusers_disable_mmap` (Settings > Model Loading), off by default,
which makes diffusers read shards into anonymous memory instead. Costs
peak RAM equal to the model size, so it is opt-in.

SD3.5-large on RX 9070 (gfx1201), same prompt and steps:
  off: no image after 120s
  on:  image in 20s

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Christopher McGee
2026-08-16 08:02:31 -04:00
parent 12bf006b00
commit a9b9de63ab
2 changed files with 2 additions and 0 deletions
+1
View File
@@ -904,6 +904,7 @@ def load_diffuser(checkpoint_info: CheckpointInfo | None = None, op='model', rev
timer.load.record("diffusers")
diffusers_load_config = {
"low_cpu_mem_usage": True,
"disable_mmap": shared.opts.diffusers_disable_mmap,
"torch_dtype": devices.dtype,
"load_connected_pipeline": True,
"safety_checker": None, # sd15 specific but we cant know ahead of time
+1
View File
@@ -93,6 +93,7 @@ def create_settings(cmd_opts):
"sd_parallel_load": OptionInfo(True, "Model load using multiple threads"),
"sd_checkpoint_autodownload": OptionInfo(True, "Model auto-download on demand"),
"stream_load": OptionInfo(False, "Model load using streams", gr.Checkbox),
"diffusers_disable_mmap": OptionInfo(False, "Model load without mmap", gr.Checkbox),
"diffusers_to_gpu": OptionInfo(False, "Model load model direct to GPU"),
"runai_streamer_diffusers": OptionInfo(False, "Diffusers load using Run:ai streamer", gr.Checkbox),
"runai_streamer_transformers": OptionInfo(False, "Transformers load using Run:ai streamer", gr.Checkbox),