mirror of
https://github.com/vladmandic/automatic
synced 2026-09-05 20:40:44 +02:00
fix(offload): budget pinned memory from available ram
The guard compared component size against half of total system memory and silently dropped to unpinned streaming above that. Total is the wrong quantity: a component that fits comfortably in free memory gets denied, and unpinned leaf streaming is slow out of proportion to its size because every per-leaf transfer pays staging cost. Budget against memory free at apply time less a reserve, and cache the verdict on the module since a granted pin lowers the same reading it derives from. A denied pin now also drops streams and falls back to block_level, since leaf plus stream is only the right shape when weights are pinned, and logs the per-step transfer volume so the cost of the slow path is visible up front.
This commit is contained in:
+17
-3
@@ -138,10 +138,24 @@ def apply_group_offload_component(module, module_name: str, main: bool, op: str
|
||||
cfg = group_offload_config(main)
|
||||
if cfg['use_stream'] and not cfg['low_cpu_mem_usage']:
|
||||
size_gb, _params = get_module_size(module)
|
||||
limit_gb = 0.5 * shared.cpu_memory # heuristic ceiling: pinned memory is non-pageable, and half of system memory must stay available to everything else
|
||||
if size_gb > limit_gb:
|
||||
pin_ok = getattr(module, 'sdnext_group_offload_pin', None)
|
||||
if pin_ok is None: # decide once per module: a granted pin moves the weights into locked memory, so re-reading available on the next apply would see it lower by the pinned size and revoke its own grant
|
||||
from modules import memstats
|
||||
avail_gb = memstats.ram_stats().get('avail', 0)
|
||||
reserve_gb = max(8.0, 0.25 * shared.cpu_memory) # pinned pages cannot be reclaimed or swapped, so a quarter of the machine, floored at 8 GB, stays pageable for the process and page cache
|
||||
limit_gb = (avail_gb - reserve_gb) if avail_gb > 0 else (0.5 * shared.cpu_memory) # budget from memory free right now; total-derived ceiling only when psutil cannot say
|
||||
pin_ok = size_gb <= limit_gb
|
||||
module.sdnext_group_offload_pin = pin_ok
|
||||
module.sdnext_group_offload_pin_limit = limit_gb
|
||||
if not pin_ok:
|
||||
# unpinned streaming degrades to per-transfer staging and leaf groups make that a per-module cost,
|
||||
# so the whole leaf+stream shape goes with the pin: few large synchronous groups instead
|
||||
cfg['low_cpu_mem_usage'] = True
|
||||
log.warning(f'Setting {op}: offload=group module={module_name} size={size_gb:.3f} limit={limit_gb:.3f} pin=dynamic memory guard')
|
||||
cfg['use_stream'] = False
|
||||
cfg['record_stream'] = False
|
||||
cfg['offload_type'] = 'block_level'
|
||||
cfg['num_blocks_per_group'] = max(4, int(shared.opts.group_offload_blocks))
|
||||
log.warning(f'Setting {op}: offload=group module={module_name} size={size_gb:.3f} limit={getattr(module, "sdnext_group_offload_pin_limit", 0):.3f} pin=denied type=block_level blocks={cfg["num_blocks_per_group"]} expect ~{size_gb:.0f} GB transferred per step')
|
||||
sig = f'{devices.device}:{main}:' + ':'.join(str(v) for v in cfg.values())
|
||||
if getattr(module, 'sdnext_group_offload_sig', None) == sig:
|
||||
return False
|
||||
|
||||
Reference in New Issue
Block a user