From 5735b8f47d59ddf413a83ceb0ed10c8dccd56bfb Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 01:44:53 +0100 Subject: [PATCH 1/9] fix(ltx): pin guidance and cross-timestep on every 2.x call path Some paths left both to pipeline defaults, which track the current upstream model: distilled picked up guidance it already bakes in, and 2.0 could land on 2.3's joint-sigma path. --- modules/ltx/ltx_process.py | 40 +++++++++++++++++++++++++------------- 1 file changed, 26 insertions(+), 14 deletions(-) diff --git a/modules/ltx/ltx_process.py b/modules/ltx/ltx_process.py index 39abe0643..5342bf597 100644 --- a/modules/ltx/ltx_process.py +++ b/modules/ltx/ltx_process.py @@ -32,11 +32,28 @@ def _prompt_tensors_to_device(*tensors): return tuple(t.to(device=devices.device) if torch.is_tensor(t) else t for t in tensors) +def identity_ltx2_guidance() -> dict: + # Named rather than omitted: pipeline defaults track the current upstream model, so a missing + # term guides a schedule that already bakes it in. + return { + 'stg_scale': 0.0, + 'modality_scale': 1.0, + 'guidance_rescale': 0.0, + 'spatio_temporal_guidance_blocks': None, + 'audio_guidance_scale': 1.0, + 'audio_stg_scale': 0.0, + 'audio_modality_scale': 1.0, + 'audio_guidance_rescale': 0.0, + } + + def _canonical_ltx2_guidance(caps) -> dict: # Four-way composition (cfg + stg + modality + rescale) from huggingface/diffusers#13217. - # Distilled bakes these into its sigma schedule; skip or we double-apply. - if caps.family != '2.x' or caps.is_distilled: + # Distilled bakes these into its sigma schedule and runs at identity. + if caps.family != '2.x': return {} + if caps.is_distilled: + return identity_ltx2_guidance() return { 'stg_scale': caps.stg_default_scale, 'modality_scale': caps.modality_default_scale, @@ -58,14 +75,7 @@ def _canonical_stage2_kwargs() -> dict: 'sigmas': list(STAGE_2_DISTILLED_SIGMA_VALUES), 'noise_scale': float(STAGE_2_DISTILLED_SIGMA_VALUES[0]), 'guidance_scale': 1.0, - 'stg_scale': 0.0, - 'modality_scale': 1.0, - 'guidance_rescale': 0.0, - 'audio_guidance_scale': 1.0, - 'audio_stg_scale': 0.0, - 'audio_modality_scale': 1.0, - 'audio_guidance_rescale': 0.0, - 'spatio_temporal_guidance_blocks': None, + **identity_ltx2_guidance(), } @@ -104,8 +114,8 @@ def _latent_pass(caps, prompt_embeds, prompt_attention_mask, negative_prompt_emb base_args['sigmas'] = list(DISTILLED_SIGMA_VALUES) base_args.pop('num_inference_steps', None) base_args.update(_canonical_ltx2_guidance(caps)) - if caps.use_cross_timestep: - base_args['use_cross_timestep'] = True + if caps.family == '2.x': + base_args['use_cross_timestep'] = caps.use_cross_timestep log.debug(f'Video: cls={shared.sd_model.__class__.__name__} op=latent_pass args_keys={list(base_args.keys())}') result = shared.sd_model(**base_args) latents = result.frames[0] if hasattr(result, 'frames') else None @@ -316,6 +326,8 @@ def run_ltx(task_id, p.task_args['sigmas'] = list(DISTILLED_SIGMA_VALUES) p.task_args.pop('num_inference_steps', None) p.task_args.update(_canonical_ltx2_guidance(caps)) + if caps.family == '2.x': + p.task_args['use_cross_timestep'] = caps.use_cross_timestep framewise = caps.family == '0.9' set_vae_params(p, framewise=framewise) @@ -499,8 +511,8 @@ def run_ltx(task_id, # Thread Stage-1 I2V init image through Stage 2 so first-frame identity survives refine. if caps.is_i2v and caps.repo_cls_name in ('LTXImageToVideoPipeline', 'LTX2ImageToVideoPipeline') and p.task_args.get('image') is not None: refine_args['image'] = p.task_args['image'] - if caps.family == '2.x' and caps.use_cross_timestep: - refine_args['use_cross_timestep'] = True + if caps.family == '2.x': + refine_args['use_cross_timestep'] = caps.use_cross_timestep # output_type='latent' skips the post-loop audio_vae + vocoder pass when audio # is unwanted; per-step audio cross-attention still runs for video conditioning. # Internal video decode is also skipped; vae_decode below picks it up. From 935c275d50cec4cb97050a70551a4dbcc62905ee Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 01:45:12 +0100 Subject: [PATCH 2/9] chore(deps): bump diffusers for ltx-2.5 Adds the Gemma 4 text encoder, the duration head and the diffusion decoder. Pipeline classes are unchanged, so 2.0 and 2.3 load as before. Image conditioning is now re-compressed to match training, at CRF 18 for 2.5 and 33 for earlier versions. --- installer.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/installer.py b/installer.py index 4d1ce2534..eda04a677 100644 --- a/installer.py +++ b/installer.py @@ -584,7 +584,7 @@ def check_diffusers(): t_start = time.time() if args.skip_all: return - target_commit = "90c0ffdc045902a3667d473d2fbfc03e8716dba9" # diffusers commit hash == 0.40.0.dev0 == 08-11-2026 + target_commit = "7564fb016dabda0c943416190fc92398c50b1b20" # diffusers commit hash == 0.40.0.dev0 == 08-11-2026 # if args.use_rocm or args.use_zluda: # sha = '043ab2520f6a19fce78e6e060a68dbc947edb9f9' # lock diffusers versions for now pkg = package_spec('diffusers') From 45fc1a0243db4371ff34652a180bc02483f61261 Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 01:53:19 +0100 Subject: [PATCH 3/9] fix(ltx): scope cached upsamplers and scheduler overrides to the run The sampler shift override looked for flow_shift, which only UniPC-style schedulers declare, so it never reached the flow-match schedulers 2.x uses. Config keys are now restored only when the run wrote them, and a cached upsampler is rebuilt once its repo or the model's VAE changes. --- modules/ltx/ltx_util.py | 83 +++++++++++++++++++++++++++-------------- 1 file changed, 56 insertions(+), 27 deletions(-) diff --git a/modules/ltx/ltx_util.py b/modules/ltx/ltx_util.py index 11afb0b1c..914c86834 100644 --- a/modules/ltx/ltx_util.py +++ b/modules/ltx/ltx_util.py @@ -37,7 +37,19 @@ def load_model(engine: str, model: str): timer.process.add('offload', t2 - t1) +def upsample_pipe_stale(upsample_pipe, upsample_repo_id) -> bool: + # bound to one repo and borrowing the model's VAE; both change with the model, and a VAE from + # an unloaded model holds meta tensors + if upsample_pipe is None: + return False + if getattr(upsample_pipe, 'sdnext_upsample_repo', None) != upsample_repo_id: + return True + return upsample_pipe.vae is not getattr(shared.sd_model, 'vae', None) + + def load_upsample(upsample_pipe, upsample_repo_id): + if upsample_pipe_stale(upsample_pipe, upsample_repo_id): + upsample_pipe = None if upsample_pipe is None: t0 = time.time() from diffusers.pipelines.ltx.pipeline_ltx_latent_upsample import LTXLatentUpsamplePipeline @@ -48,14 +60,17 @@ def load_upsample(upsample_pipe, upsample_repo_id): cache_dir=shared.opts.hfcache_dir, torch_dtype=devices.dtype, ) + upsample_pipe.sdnext_upsample_repo = upsample_repo_id t1 = time.time() timer.process.add('load', t1 - t0) return upsample_pipe -def load_upsample_2x(upsample_pipe, upsample_repo_id): +def load_upsample_2x(upsample_pipe, upsample_repo_id, variant: str = '2.x'): # 2.x ships the upsampler as a bare nn.Module in a subfolder; no from_pretrained on the # pipeline wrapper, so we load the model + construct the pipeline manually. + if upsample_pipe_stale(upsample_pipe, upsample_repo_id): + upsample_pipe = None if upsample_pipe is None: t0 = time.time() from diffusers.pipelines.ltx2.pipeline_ltx2_latent_upsample import LTX2LatentUpsamplePipeline @@ -74,40 +89,61 @@ def load_upsample_2x(upsample_pipe, upsample_repo_id): ) # Synthetic checkpoint_info gives this pipe its own OffloadHook cache slot, so routing # it through apply_balanced_offload does not invalidate the main pipe's module map - # (sd_offload.py:488 keys on sd_checkpoint_info.name). - upsample_pipe.sd_checkpoint_info = sd_checkpoint.CheckpointInfo('ltx-upsampler-2.x') + # (sd_offload keys on sd_checkpoint_info.name). Variant is in the name since each loads + # different weights and the slot also names the disk offload folder. + upsample_pipe.sdnext_upsample_repo = upsample_repo_id + upsample_pipe.sd_checkpoint_info = sd_checkpoint.CheckpointInfo(f'ltx-upsampler-{variant}') t1 = time.time() timer.process.add('load', t1 - t0) return upsample_pipe +def scheduler_shift_key(scheduler) -> str | None: + # UniPC and its relatives call the static shift flow_shift; flow-match schedulers call it shift. + config = getattr(scheduler, 'config', None) + if config is None: + return None + for key in ('flow_shift', 'shift'): + if hasattr(config, key): + return key + return None + + @contextmanager -def ltx_scheduler_opts(sd_model, *, dynamic_shift=None, sampler_shift=None): - # Run-scoped override of shared.opts scheduler settings and scheduler.config. Snapshots - # five pieces of state (shared.opts dynamic_shift + shift, scheduler object, default_scheduler - # snapshot, and scheduler.config use_dynamic_shifting + flow_shift) and restores every one on - # exit. Keeps run-specific sampler settings out of config.json and prevents default_scheduler - # from getting clobbered by a deepcopy of the mutated scheduler at video_load.py:171. The - # scheduler-object restore matters for Stage 2 refine, which swaps the scheduler entirely. +def ltx_scheduler_opts(sd_model, *, dynamic_shift=None, sampler_shift=None, shift_terminal=None): + # Run-scoped override of shared.opts scheduler settings and scheduler.config, restored on every + # exit path. Keeps run settings out of config.json, protects default_scheduler from a deepcopy + # of the mutated scheduler, and restores the scheduler object that Stage 2 refine swaps out. orig_dynamic_shift = shared.opts.schedulers_dynamic_shift orig_sampler_shift = shared.opts.schedulers_shift orig_scheduler = sd_model.scheduler orig_default_scheduler = getattr(sd_model, 'default_scheduler', None) - orig_use_dynamic_shifting = getattr(orig_scheduler.config, 'use_dynamic_shifting', None) if hasattr(orig_scheduler, 'config') else None - orig_flow_shift = getattr(orig_scheduler.config, 'flow_shift', None) if hasattr(orig_scheduler, 'config') else None + restore = {} + + def write_config(values: dict): + scheduler = getattr(sd_model, 'scheduler', None) + if not values or scheduler is None or not hasattr(scheduler, 'config') or not hasattr(scheduler, 'register_to_config'): + return + for key, value in values.items(): + setattr(scheduler.config, key, value) + scheduler.register_to_config(**values) + + def override_config(key, value): + # only written keys are restored, so a key stored as None returns to None + if key is None or value is None or not hasattr(getattr(orig_scheduler, 'config', None), key): + return + restore[key] = getattr(orig_scheduler.config, key) + write_config({key: value}) try: if dynamic_shift is not None: shared.opts.data['schedulers_dynamic_shift'] = dynamic_shift if sampler_shift is not None: shared.opts.data['schedulers_shift'] = sampler_shift - if hasattr(sd_model, 'scheduler') and hasattr(sd_model.scheduler, 'config') and hasattr(sd_model.scheduler, 'register_to_config'): - if dynamic_shift is not None and hasattr(sd_model.scheduler.config, 'use_dynamic_shifting'): - sd_model.scheduler.config.use_dynamic_shifting = dynamic_shift - sd_model.scheduler.register_to_config(use_dynamic_shifting=dynamic_shift) - if sampler_shift is not None and sampler_shift >= 0 and hasattr(sd_model.scheduler.config, 'flow_shift'): - sd_model.scheduler.config.flow_shift = sampler_shift - sd_model.scheduler.register_to_config(flow_shift=sampler_shift) + override_config('use_dynamic_shifting', dynamic_shift) + if sampler_shift is not None and sampler_shift >= 0: + override_config(scheduler_shift_key(orig_scheduler), sampler_shift) + override_config('shift_terminal', shift_terminal) yield finally: shared.opts.data['schedulers_dynamic_shift'] = orig_dynamic_shift @@ -116,14 +152,7 @@ def ltx_scheduler_opts(sd_model, *, dynamic_shift=None, sampler_shift=None): sd_model.scheduler = orig_scheduler if orig_default_scheduler is not None and sd_model.default_scheduler is not orig_default_scheduler: sd_model.default_scheduler = orig_default_scheduler - if hasattr(sd_model.scheduler, 'config') and hasattr(sd_model.scheduler, 'register_to_config'): - if orig_use_dynamic_shifting is not None and hasattr(sd_model.scheduler.config, 'use_dynamic_shifting'): - sd_model.scheduler.config.use_dynamic_shifting = orig_use_dynamic_shifting - sd_model.scheduler.register_to_config(use_dynamic_shifting=orig_use_dynamic_shifting) - if orig_flow_shift is not None and hasattr(sd_model.scheduler.config, 'flow_shift'): - sd_model.scheduler.config.flow_shift = orig_flow_shift - sd_model.scheduler.register_to_config(flow_shift=orig_flow_shift) - # log.debug(f'LTX: scheduler/opts restored dynamic_shift={orig_dynamic_shift} sampler_shift={orig_sampler_shift}') + write_config(restore) def _condition_cls(family: str): From 780495ff2e7686952e5f6a7b5a32bccde78c7404 Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 01:53:19 +0100 Subject: [PATCH 4/9] feat(video): allow a transformer folder to load into another argument The subfolder name doubles as the pipeline argument, so a repo shipping a second transformer beside the default one had no way to load it: the argument was dropped and the default weights stayed. dit_kwarg names the argument when it differs from the folder. --- modules/video_models/models_def.py | 1 + modules/video_models/video_load.py | 15 +++++++++------ 2 files changed, 10 insertions(+), 6 deletions(-) diff --git a/modules/video_models/models_def.py b/modules/video_models/models_def.py index 42b041a07..107dafab1 100644 --- a/modules/video_models/models_def.py +++ b/modules/video_models/models_def.py @@ -16,6 +16,7 @@ class Model: dit: str = None dit_cls: classmethod = None dit_folder: str = 'transformer' + dit_kwarg: str = None # pipeline argument the folder loads into, when the two differ dit_revision: str = None te: str = None te_cls: classmethod = None diff --git a/modules/video_models/video_load.py b/modules/video_models/video_load.py index 4e7415ae7..ff1a981b0 100644 --- a/modules/video_models/video_load.py +++ b/modules/video_models/video_load.py @@ -143,12 +143,13 @@ def load_model(selected: models_def.Model): # transformer if selected.dit_cls is not None: try: - def load_dit_folder(dit_folder): - if dit_folder is not None and dit_folder not in kwargs: + def load_dit_folder(dit_folder, dit_kwarg=None): + dit_kwarg = dit_kwarg or dit_folder # ltx-2.5 keeps its dev transformer in transformer_full + if dit_folder is not None and dit_kwarg not in kwargs: # get a new quant arg on every loop to prevent the quant config classes getting entangled load_args, quant_args = model_quant.get_dit_args({}, module='Model', device_map=True) - log.debug(f'Load video: module=transformer repo="{selected.dit or selected.repo}" module="{dit_folder}" folder="{dit_folder}" cls={selected.dit_cls.__name__} quant={model_quant.get_quant_type(quant_args)} loader={_loader("diffusers")}') - kwargs[dit_folder] = selected.dit_cls.from_pretrained( + log.debug(f'Load video: module=transformer repo="{selected.dit or selected.repo}" module="{dit_kwarg}" folder="{dit_folder}" cls={selected.dit_cls.__name__} quant={model_quant.get_quant_type(quant_args)} loader={_loader("diffusers")}') + kwargs[dit_kwarg] = selected.dit_cls.from_pretrained( pretrained_model_name_or_path=selected.dit or selected.repo, subfolder=dit_folder, revision=selected.dit_revision or selected.repo_revision, @@ -158,15 +159,17 @@ def load_model(selected: models_def.Model): **offline_args, ) else: - log.debug(f'Load video: module=transformer repo="{selected.dit or selected.repo}" module="{dit_folder}" folder="{dit_folder}" cls={selected.dit_cls.__name__} loader={_loader("diffusers")} skip') + log.debug(f'Load video: module=transformer repo="{selected.dit or selected.repo}" module="{dit_kwarg}" folder="{dit_folder}" cls={selected.dit_cls.__name__} loader={_loader("diffusers")} skip') if selected.dit_folder is None: selected.dit_folder = ['transformer'] if isinstance(selected.dit_folder, list) or isinstance(selected.dit_folder, tuple): + if selected.dit_kwarg is not None: + log.warning(f'Load video: model="{selected.name}" dit_kwarg unsupported with multiple folders') for dit_folder in selected.dit_folder: # wan a14b has transformer and transformer_2 load_dit_folder(dit_folder) else: - load_dit_folder(selected.dit_folder) + load_dit_folder(selected.dit_folder, selected.dit_kwarg) except Exception as e: log.error(f'video load: module=transformer cls={selected.dit_cls.__name__} {e}') errors.display(e, 'video') From f01e752b062bb3343ea720b7433eb46811ff29cc Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 02:03:06 +0100 Subject: [PATCH 5/9] fix(video): take the audio rate from the loaded vocoder Pipelines rarely report a sample rate, so the save path fell back to 24000. LTX-2.3 and 2.5 run at 48k, and muxing at half the rate drops the track an octave. The rate now comes from the vocoder, as the LTX tab already did. --- modules/video_models/video_run.py | 2 +- modules/video_models/video_save.py | 10 ++++++++++ 2 files changed, 11 insertions(+), 1 deletion(-) diff --git a/modules/video_models/video_run.py b/modules/video_models/video_run.py index 6444607eb..91e683e91 100644 --- a/modules/video_models/video_run.py +++ b/modules/video_models/video_run.py @@ -326,7 +326,7 @@ def run(selected: models_def.Model, *, p=p, pixels=pixels, audio=waveform, - aac_sample_rate=getattr(p, 'audio_sampling_rate', None) or 24000, + aac_sample_rate=video_save.get_audio_rate(p), binary=processed.bytes, mp4_fps=save_fps, mp4_codec=mp4_codec, diff --git a/modules/video_models/video_save.py b/modules/video_models/video_save.py index b14274ee6..323b67403 100644 --- a/modules/video_models/video_save.py +++ b/modules/video_models/video_save.py @@ -11,6 +11,16 @@ from modules.logger import log from modules.video_models.video_utils import check_av +def get_audio_rate(p=None, default: int = 24000) -> int: + # pipeline output wins when it reports a rate, else the loaded vocoder: LTX-2.0 runs at 24k, + # 2.3 and 2.5 at 48k, and muxing at the wrong rate shifts the pitch + rate = getattr(p, 'audio_sampling_rate', None) if p is not None else None + if not rate: + vocoder = getattr(shared.sd_model, 'vocoder', None) + rate = getattr(getattr(vocoder, 'config', None), 'output_sampling_rate', None) + return int(rate) if rate else default + + def get_video_filename(p:processing.StableDiffusionProcessingVideo): from modules.image.namegen import FilenameGenerator from modules.paths import resolve_output_path From 0d1882eca4ed047c882f0ae17e6953e6224ca88b Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 02:03:06 +0100 Subject: [PATCH 6/9] fix(video): keep text encoder dedup out of the registry rows The shared repo was written back onto the registry row, a module-level singleton, so turning the setting off left the row pointing at the shared copy for the rest of the session. It is chosen into locals instead. --- modules/video_models/video_load.py | 52 ++++++++++++------------------ 1 file changed, 21 insertions(+), 31 deletions(-) diff --git a/modules/video_models/video_load.py b/modules/video_models/video_load.py index ff1a981b0..d1c94aca9 100644 --- a/modules/video_models/video_load.py +++ b/modules/video_models/video_load.py @@ -98,39 +98,29 @@ def load_model(selected: models_def.Model): try: load_args, quant_args = model_quant.get_dit_args({}, module='TE', device_map=True) - # loader deduplication of text-encoder models - if selected.te_cls.__name__ == 'T5EncoderModel' and shared.opts.te_shared_te: - selected.te = 'Disty0/t5-xxl' - selected.te_folder = '' - selected.te_revision = None - if selected.te_cls.__name__ == 'UMT5EncoderModel' and shared.opts.te_shared_te: - if 'SDNQ' in selected.name: - selected.te = 'Disty0/Wan2.2-T2V-A14B-SDNQ-uint4-svd-r32' - else: - selected.te = 'Wan-AI/Wan2.2-TI2V-5B-Diffusers' - selected.te_folder = 'text_encoder' - selected.te_revision = None - if selected.te_cls.__name__ == 'LlamaModel' and shared.opts.te_shared_te: - selected.te = 'hunyuanvideo-community/HunyuanVideo' - selected.te_folder = 'text_encoder' - selected.te_revision = None - if selected.te_cls.__name__ == 'Qwen2_5_VLForConditionalGeneration' and shared.opts.te_shared_te: - selected.te = 'ai-forever/Kandinsky-5.0-T2V-Lite-sft-5s-Diffusers' - selected.te_folder = 'text_encoder' - selected.te_revision = None - if selected.te_cls.__name__ == 'Gemma3ForConditionalGeneration' and shared.opts.te_shared_te: - if 'SDNQ' in selected.name: - selected.te = 'OzzyGT/LTX-2.3-sdnq-dynamic-int4' - else: - selected.te = 'OzzyGT/LTX-2.3' - selected.te_folder = 'text_encoder' - selected.te_revision = None + # loader deduplication of text-encoder models: picked per load, not written back onto + # the registry row where it would outlive the setting + te_repo, te_folder, te_revision = selected.te, selected.te_folder, selected.te_revision + if shared.opts.te_shared_te: + te_cls_name = selected.te_cls.__name__ + if te_cls_name == 'T5EncoderModel': + te_repo, te_folder, te_revision = 'Disty0/t5-xxl', '', None + elif te_cls_name == 'UMT5EncoderModel': + te_repo = 'Disty0/Wan2.2-T2V-A14B-SDNQ-uint4-svd-r32' if 'SDNQ' in selected.name else 'Wan-AI/Wan2.2-TI2V-5B-Diffusers' + te_folder, te_revision = 'text_encoder', None + elif te_cls_name == 'LlamaModel': + te_repo, te_folder, te_revision = 'hunyuanvideo-community/HunyuanVideo', 'text_encoder', None + elif te_cls_name == 'Qwen2_5_VLForConditionalGeneration': + te_repo, te_folder, te_revision = 'ai-forever/Kandinsky-5.0-T2V-Lite-sft-5s-Diffusers', 'text_encoder', None + elif te_cls_name == 'Gemma3ForConditionalGeneration': + te_repo = 'OzzyGT/LTX-2.3-sdnq-dynamic-int4' if 'SDNQ' in selected.name else 'OzzyGT/LTX-2.3' + te_folder, te_revision = 'text_encoder', None - log.debug(f'Load video: module=te repo="{selected.te or selected.repo}" folder="{selected.te_folder}" cls={selected.te_cls.__name__} quant={model_quant.get_quant_type(quant_args)} loader={_loader("transformers")}') + log.debug(f'Load video: module=te repo="{te_repo or selected.repo}" folder="{te_folder}" cls={selected.te_cls.__name__} quant={model_quant.get_quant_type(quant_args)} loader={_loader("transformers")}') kwargs["text_encoder"] = selected.te_cls.from_pretrained( - pretrained_model_name_or_path=selected.te or selected.repo, - subfolder=selected.te_folder, - revision=selected.te_revision or selected.repo_revision, + pretrained_model_name_or_path=te_repo or selected.repo, + subfolder=te_folder, + revision=te_revision or selected.repo_revision, cache_dir=shared.opts.hfcache_dir, **load_args, **quant_args, From 5814d5c4b3cd2db8ca31aec5983a7729de121dbb Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 02:03:18 +0100 Subject: [PATCH 7/9] feat(ltx): add the ltx-2.5 model family 2.5 reuses the LTX-2 pipeline classes, so it is described through the capability table: Gemma 4 text encoder, cross-timestep conditioning, and the upsampler and stage 2 LoRA that now ship inside the model repo. The repo carries the distilled scheduler config, so Dev rows restore the terminal shift, and the Dev transformer sits in transformer_full. Distilled rows stop forcing dynamic shifting on, which remapped their sigma schedule. Auto duration hands the clip length to the duration head. --- CHANGELOG.md | 14 ++++++++ modules/ltx/ltx_capabilities.py | 40 ++++++++++++++++----- modules/ltx/ltx_process.py | 40 ++++++++++++--------- modules/ltx/ltx_ui.py | 8 ++++- modules/video_models/models_def.py | 46 +++++++++++++++++++++++++ modules/video_models/video_overrides.py | 21 +++++++---- ui/locale/locale_en.json | 9 ++--- 7 files changed, 140 insertions(+), 38 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 681cc1331..47e1f166b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,14 @@ - [MiniMax H3](https://huggingface.co/MiniMaxAI/MiniMax-H3) in *base* and *ref* variants MiniMax-H3 is an amazing, but absolutely massive at 32B text-encoder and 33B transformer video model for details, see [MiniMax wiki page](wiki/MiniMax) + - [LTX-2.5](https://huggingface.co/Lightricks/LTX-2.5) from Lightricks + 22B joint audio and video generation in *Distilled* and *Dev* variants, + each as *text-to-video*, *image-to-video* and *conditioning* rows + paired with a *Gemma 4* text encoder and a duration head: + tick *auto duration* and the clip length is predicted from the prompt instead of set by hand + *note*: LTX-2.5 is a [gated model](https://vladmandic.github.io/sdnext-docs/Gated/) + *note*: image conditioning is now re-compressed to match what the models were trained on, + at CRF 18 for LTX-2.5 and 33 for earlier LTX-2.x versions - **Detailer**: Pretty much *detailer.next* :) Detailer detection models were traditionally *YOLO* models, but now we can also use: - [Facebook-SAM3](https://huggingface.co/facebook/sam3) hybrid promptable concept segmentation and detection network @@ -43,6 +51,12 @@ - improve pipeline detection for non-cached models - cleanup alt offload codepaths - hf progress bars + - ltx: send the guidance stack and cross-timestep on every 2.x call path + - ltx: distilled variants no longer force dynamic shifting on, which remapped their sigma schedule + - ltx: sampler shift now reaches flow-match schedulers + - ltx: reload the latent upsampler when the model or its repo changes + - video: take the audio sample rate from the loaded vocoder + - video: keep the shared text encoder out of the registry rows ## Update for 2026-08-07 diff --git a/modules/ltx/ltx_capabilities.py b/modules/ltx/ltx_capabilities.py index 0d1909580..662d8c6d9 100644 --- a/modules/ltx/ltx_capabilities.py +++ b/modules/ltx/ltx_capabilities.py @@ -9,7 +9,7 @@ class LTXCaps: name: str repo_cls_name: str family: str # '0.9' or '2.x' - variant: str # '0.9', '2.0', '2.3' (finer-grained sub-variant) + variant: str # '0.9', '2.0', '2.3', '2.5' (finer-grained sub-variant) is_distilled: bool is_i2v: bool supports_input_media: bool @@ -41,6 +41,13 @@ class LTXCaps: # for refine; same-res refine oversaturates. Condition variants rebuild conditions per stage. supports_two_stage_refine: bool = False stage2_dev_lora_repo: Optional[str] = None + # 2.5 keeps the stage 2 LoRA in the model repo; 2.0 and 2.3 each have their own + stage2_dev_lora_weight: Optional[str] = None + # tied to the family VAE: the wrong one drifts per-channel latent statistics + upsample_repo: Optional[str] = None + # 2.5 ships the distilled scheduler config, so its Dev rows restore the terminal shift + scheduler_shift_terminal: Optional[float] = None + supports_auto_duration: bool = False CONDITION_CLASSES = {'LTXConditionPipeline', 'LTX2ConditionPipeline'} @@ -84,7 +91,12 @@ def get_caps(model_name: str) -> Optional[LTXCaps]: family = '2.x' if is_ltx2 else '0.9' # 2.x sub-variant detection: unknown 2.x mirrors fall through to '2.0' (conservative default). if is_ltx2: - variant = '2.3' if '2.3' in model_name else '2.0' + if '2.5' in model_name: + variant = '2.5' + elif '2.3' in model_name: + variant = '2.3' + else: + variant = '2.0' else: variant = '0.9' is_distilled = 'Distilled' in model_name @@ -106,11 +118,12 @@ def get_caps(model_name: str) -> Optional[LTXCaps]: supports_stg=is_ltx2, supports_audio=is_ltx2, supports_frame_rate_kwarg=is_ltx2, - use_cross_timestep=(variant == '2.3'), + use_cross_timestep=variant in ('2.3', '2.5'), default_cfg=3.0, default_steps=30 if is_ltx2 else 50, default_sampler_shift=-1.0, - default_dynamic_shift=is_ltx2, + # distilled ships use_dynamic_shifting=False and runs explicit sigmas, which shifting remaps + default_dynamic_shift=is_ltx2 and not is_distilled, default_width=768, default_height=512, default_frames=121 if is_ltx2 else 161, @@ -122,7 +135,10 @@ def get_caps(model_name: str) -> Optional[LTXCaps]: caps.default_steps = 8 if is_ltx2 and not is_distilled: - if variant == '2.3': + if variant == '2.5': + caps.stage2_dev_lora_repo = 'Lightricks/LTX-2.5-Diffusers' + caps.stage2_dev_lora_weight = 'ltx-2.5-22b-distilled-lora-450-bf16.safetensors' + elif variant == '2.3': caps.stage2_dev_lora_repo = 'CalamitousFelicitousness/LTX-2.3-distilled-lora-384-Diffusers' elif variant == '2.0': caps.stage2_dev_lora_repo = 'CalamitousFelicitousness/LTX-2.0-distilled-lora-384-Diffusers' @@ -130,16 +146,22 @@ def get_caps(model_name: str) -> Optional[LTXCaps]: caps.supports_two_stage_refine = is_ltx2 if is_ltx2: - if variant == '2.3': + if variant == '2.5': + caps.upsample_repo = 'Lightricks/LTX-2.5-Diffusers' + caps.stg_default_blocks = [28] + caps.supports_auto_duration = True + elif variant == '2.3': + caps.upsample_repo = 'CalamitousFelicitousness/LTX-2.3-Spatial-Upsampler-x2-1.1-Diffusers' caps.stg_default_blocks = [28] - elif variant == '2.0': - caps.stg_default_blocks = [29] else: - caps.stg_default_blocks = [28] + caps.upsample_repo = 'Lightricks/LTX-2' + caps.stg_default_blocks = [29] if not is_distilled: # canonical T2V composition from huggingface/diffusers#13217 caps.stg_default_scale = 1.0 caps.modality_default_scale = 3.0 caps.guidance_rescale_default = 0.7 + if variant == '2.5': + caps.scheduler_shift_terminal = 0.1 return caps diff --git a/modules/ltx/ltx_process.py b/modules/ltx/ltx_process.py index 5342bf597..c684dcee2 100644 --- a/modules/ltx/ltx_process.py +++ b/modules/ltx/ltx_process.py @@ -12,16 +12,12 @@ from modules.ltx.ltx_util import get_bucket, get_frames, load_model, load_upsamp apply_ltx_diffusers_patch() from modules.processing_callbacks import diffusers_callback from modules.video_models.video_vae import set_vae_params -from modules.video_models.video_save import save_video +from modules.video_models.video_save import save_video, get_audio_rate from modules.video_models.video_utils import check_av debug = log.trace if os.environ.get('SD_VIDEO_DEBUG', None) is not None else lambda *args, **kwargs: None upsample_repo_id_09 = 'a-r-r-o-w/LTX-Video-0.9.7-Latent-Spatial-Upsampler-diffusers' -# Upsampler weights are tied to the family VAE; using the wrong one preserves structure -# but drifts per-channel latent statistics (decodes desaturated / crushed contrast). -upsample_repo_id_20 = 'Lightricks/LTX-2' -upsample_repo_id_23 = 'CalamitousFelicitousness/LTX-2.3-Spatial-Upsampler-x2-1.1-Diffusers' upsample_pipe = None upsample_pipe_2x = None @@ -93,7 +89,7 @@ def _latent_pass(caps, prompt_embeds, prompt_attention_mask, negative_prompt_emb 'negative_prompt_attention_mask': negative_prompt_attention_mask, 'width': get_bucket(width), 'height': get_bucket(height), - 'num_frames': get_frames(frames), + 'num_frames': get_frames(frames) if frames is not None else None, # None defers to the duration head 'num_inference_steps': steps, 'generator': get_generator(seed), 'callback_on_step_end': diffusers_callback, @@ -131,6 +127,7 @@ def run_ltx(task_id, width: int, height: int, frames: int, + auto_duration: bool, steps: int, sampler_index: int, guidance_scale: float, @@ -199,6 +196,10 @@ def run_ltx(task_id, yield from abort(f'Video: cls={shared.sd_model.__class__.__name__} selected model is not LTX', ok=True) return + auto_frames = bool(auto_duration) and caps.supports_auto_duration + if auto_duration and not auto_frames: + log.warning(f'LTX: model="{model}" auto duration unsupported, using frames={get_frames(frames)}') + # Lightricks TI2VidTwoStagesPipeline: Stage 1 at half-res, 2x upsample, Stage 2 refine at target. # Auto-couple when the user picks Refine but not Upsample. Both Dev and Distilled refine paths # expect upsampled latents; same-res refine on Distilled produces oversaturation. Condition @@ -328,14 +329,15 @@ def run_ltx(task_id, p.task_args.update(_canonical_ltx2_guidance(caps)) if caps.family == '2.x': p.task_args['use_cross_timestep'] = caps.use_cross_timestep + if auto_frames: + p.task_args['num_frames'] = None framewise = caps.family == '0.9' set_vae_params(p, framewise=framewise) # Scheduler + shared.opts mutation is wrapped in ltx_scheduler_opts so restore runs on - # every exit path (normal return, abort, interrupt, Stage 2 scheduler swap). See the - # helper's docstring for the five pieces of state it snapshots. - with ltx_scheduler_opts(shared.sd_model, dynamic_shift=dynamic_shift, sampler_shift=sampler_shift): + # every exit path (normal return, abort, interrupt, Stage 2 scheduler swap). + with ltx_scheduler_opts(shared.sd_model, dynamic_shift=dynamic_shift, sampler_shift=sampler_shift, shift_terminal=caps.scheduler_shift_terminal): if selected is not None: video_overrides.set_overrides(p, selected) @@ -382,7 +384,7 @@ def run_ltx(task_id, negative_prompt_attention_mask=negative_prompt_attention_mask, width=base_w, height=base_h, - frames=frames, + frames=None if auto_frames else frames, steps=steps, guidance_scale=p.cfg_scale, mp4_fps=mp4_fps, @@ -391,6 +393,11 @@ def run_ltx(task_id, seed=p.seed, image=p.task_args.get('image'), ) + if auto_frames and torch.is_tensor(latents): + # upsample and refine take the realized length; re-predicting would drift + frames = (latents.shape[-3] - 1) * getattr(shared.sd_model, 'vae_temporal_compression_ratio', 8) + 1 + p.frames = frames + log.debug(f'LTX: auto duration frames={frames}') else: processed = processing.process_images(p) if processed is None or processed.images is None or len(processed.images) == 0: @@ -451,8 +458,7 @@ def run_ltx(task_id, upsample_pipe = sd_models.apply_balanced_offload(upsample_pipe, exclude=upsample_exclude, silent=True) else: global upsample_pipe_2x # pylint: disable=global-statement - upsample_repo = upsample_repo_id_23 if caps.variant == '2.3' else upsample_repo_id_20 - upsample_pipe_2x = load_upsample_2x(upsample_pipe_2x, upsample_repo) + upsample_pipe_2x = load_upsample_2x(upsample_pipe_2x, caps.upsample_repo, caps.variant) upsample_pipe_2x = sd_models.apply_balanced_offload(upsample_pipe_2x, exclude=upsample_exclude, silent=True) # 2.x base pass returns denormalized latents; latents_normalized=False tells the # upsampler "already raw, do not denormalize again". @@ -534,12 +540,15 @@ def run_ltx(task_id, shift_terminal=None, ) if caps.supports_canonical_stage2: - log.debug(f'LTX: stage=2 distilled=LoRA repo={caps.stage2_dev_lora_repo}') + log.debug(f'LTX: stage=2 distilled=LoRA repo={caps.stage2_dev_lora_repo} weight={caps.stage2_dev_lora_weight}') offline_args = {'local_files_only': True} if shared.opts.offline_mode else {} + # 2.5 keeps the LoRA in the model repo, so the file has to be named + lora_args ={'weight_name': caps.stage2_dev_lora_weight} if caps.stage2_dev_lora_weight is not None else {} shared.sd_model.load_lora_weights( caps.stage2_dev_lora_repo, adapter_name=STAGE2_DEV_LORA_ADAPTER, cache_dir=shared.opts.hfcache_dir, + **lora_args, **offline_args, ) shared.sd_model.set_adapters([STAGE2_DEV_LORA_ADAPTER], [1.0]) @@ -628,10 +637,7 @@ def run_ltx(task_id, if not audio_enable: audio = None - try: - aac_sample_rate = shared.sd_model.vocoder.config.output_sampling_rate - except Exception: - aac_sample_rate = 24000 + aac_sample_rate = get_audio_rate(p) if mp4_interpolate > 0 and pixels is not None: p.video_interpolate = mp4_interpolate diff --git a/modules/ltx/ltx_ui.py b/modules/ltx/ltx_ui.py index 3d381933f..550b79686 100644 --- a/modules/ltx/ltx_ui.py +++ b/modules/ltx/ltx_ui.py @@ -29,6 +29,7 @@ def _model_change(model_name: str): gr.update(interactive=False), # decode_timestep gr.update(interactive=False), # image_cond_noise_scale gr.update(visible=False), # audio_accordion + gr.update(visible=False, value=False), # auto_duration ) # 2.x refine runs fixed canonical schedules; refine_strength only feeds 0.9.x LTXConditionPipeline. refine_strength_interactive = caps.family == '0.9' @@ -36,6 +37,7 @@ def _model_change(model_name: str): # Distilled T2V/I2V). auto_refine_upsample at ltx_process.py:179 couples the stages once Refine # is on. Condition variants are excluded by supports_two_stage_refine. refine_default = caps.supports_two_stage_refine + auto_duration_update =gr.update(visible=True) if caps.supports_auto_duration else gr.update(visible=False, value=False) return ( gr.update(visible=caps.supports_input_media), gr.update(visible=caps.supports_multi_condition), @@ -52,6 +54,7 @@ def _model_change(model_name: str): gr.update(interactive=caps.supports_decode_timestep), gr.update(interactive=caps.supports_image_cond_noise_scale), gr.update(visible=caps.supports_audio), + auto_duration_update, ) @@ -71,6 +74,8 @@ def create_ui(prompt, negative, styles, overrides, script_inputs, mp4_fps, mp4_i seed = gr.Number(label='LTX seed', value=-1, elem_id='ltx_seed', container=True) random_seed = ToolButton(ui_symbols.random, elem_id='ltx_seed_random') random_seed.click(fn=lambda: -1, show_progress='hidden', inputs=[], outputs=[seed]) + with gr.Row(): + auto_duration = gr.Checkbox(label='LTX auto duration', value=False, elem_id='ltx_auto_duration', visible=False) input_media_accordion = gr.Accordion(open=False, label="Input media", elem_id='ltx_input_media_accordion', visible=False) with input_media_accordion: ltx_init_image = gr.Image(label='Image', elem_id='ltx_init_image', type='pil', image_mode='RGB', width=256, height=256) @@ -145,6 +150,7 @@ def create_ui(prompt, negative, styles, overrides, script_inputs, mp4_fps, mp4_i decode_timestep, image_cond_noise_scale, audio_accordion, + auto_duration, ], ) @@ -155,7 +161,7 @@ def create_ui(prompt, negative, styles, overrides, script_inputs, mp4_fps, mp4_i video_inputs = [ model, prompt, negative, styles, - width, height, frames, + width, height, frames, auto_duration, steps, sampler_index, guidance_scale, sampler_shift, dynamic_shift, seed, diff --git a/modules/video_models/models_def.py b/modules/video_models/models_def.py index 107dafab1..12dcdc27d 100644 --- a/modules/video_models/models_def.py +++ b/modules/video_models/models_def.py @@ -149,6 +149,52 @@ try: 'LTX Video': [ Model(name='None'), + Model(name='─────── LTX-2.5 ───────'), + Model(name='─── Distilled ───'), + Model(name='LTXVideo 2.5 22B T2V Distilled', + url='https://huggingface.co/Lightricks/LTX-2.5', + repo='Lightricks/LTX-2.5-Diffusers', + repo_cls='LTX2Pipeline', + te_cls='Gemma4UnifiedForConditionalGeneration', + dit_cls='LTX2VideoTransformer3DModel'), + Model(name='LTXVideo 2.5 22B I2V Distilled', + url='https://huggingface.co/Lightricks/LTX-2.5', + repo='Lightricks/LTX-2.5-Diffusers', + repo_cls='LTX2ImageToVideoPipeline', + te_cls='Gemma4UnifiedForConditionalGeneration', + dit_cls='LTX2VideoTransformer3DModel'), + Model(name='LTXVideo 2.5 22B Condition Distilled', + url='https://huggingface.co/Lightricks/LTX-2.5', + repo='Lightricks/LTX-2.5-Diffusers', + repo_cls='LTX2ConditionPipeline', + te_cls='Gemma4UnifiedForConditionalGeneration', + dit_cls='LTX2VideoTransformer3DModel'), + Model(name='─── Dev ───'), # transformer_full is the guided model, transformer the distilled one + Model(name='LTXVideo 2.5 22B T2V Dev', + url='https://huggingface.co/Lightricks/LTX-2.5', + repo='Lightricks/LTX-2.5-Diffusers', + repo_cls='LTX2Pipeline', + te_cls='Gemma4UnifiedForConditionalGeneration', + dit_cls='LTX2VideoTransformer3DModel', + dit_folder='transformer_full', + dit_kwarg='transformer'), + Model(name='LTXVideo 2.5 22B I2V Dev', + url='https://huggingface.co/Lightricks/LTX-2.5', + repo='Lightricks/LTX-2.5-Diffusers', + repo_cls='LTX2ImageToVideoPipeline', + te_cls='Gemma4UnifiedForConditionalGeneration', + dit_cls='LTX2VideoTransformer3DModel', + dit_folder='transformer_full', + dit_kwarg='transformer'), + Model(name='LTXVideo 2.5 22B Condition Dev', + url='https://huggingface.co/Lightricks/LTX-2.5', + repo='Lightricks/LTX-2.5-Diffusers', + repo_cls='LTX2ConditionPipeline', + te_cls='Gemma4UnifiedForConditionalGeneration', + dit_cls='LTX2VideoTransformer3DModel', + dit_folder='transformer_full', + dit_kwarg='transformer'), + Model(name='─────── LTX-2.3 v1.1 ───────'), Model(name='LTXVideo 2.3-1.1 22B T2V Distilled', url='https://huggingface.co/Lightricks/LTX-2.3', diff --git a/modules/video_models/video_overrides.py b/modules/video_models/video_overrides.py index 69afe7fb4..1964bdc59 100644 --- a/modules/video_models/video_overrides.py +++ b/modules/video_models/video_overrides.py @@ -17,17 +17,24 @@ def load_override(selected: Model, **load_args): # LTX if 'LTXVideo 0.9.5 I2V' in selected.name: kwargs['vae'] = diffusers.AutoencoderKLLTXVideo.from_pretrained(selected.repo, subfolder="vae", torch_dtype=torch.float32, cache_dir=shared.opts.hfcache_dir, **load_args) - # OzzyGT LTX-2.3 mirrors pack connectors/ twice by design: sharded (*-00001-of-0000N + - # .index.json) and unsharded diffusion_pytorch_model.safetensors of the byte-identical - # weights. snapshot_download faithfully fetches both; diffusers' component loader picks - # sharded when the index is present. ignore_patterns skips the ~6.3 GB unsharded copy - # without reaching for a cleaner upstream mirror. + # OzzyGT LTX-2.3 mirrors and the LTX-2.5 repo pack connectors/ twice by design: sharded + # (*-00001-of-0000N + .index.json) and unsharded diffusion_pytorch_model.safetensors of the + # byte-identical weights. snapshot_download faithfully fetches both; diffusers' component + # loader picks sharded when the index is present. ignore_patterns skips the ~6.3 GB unsharded + # copy without reaching for a cleaner upstream mirror. ltx2_redundant_connector_repos = { 'OzzyGT/LTX-2.3', 'OzzyGT/LTX-2.3-sdnq-dynamic-int4', } - if selected.repo in ltx2_redundant_connector_repos: - kwargs['ignore_patterns'] = ['connectors/diffusion_pytorch_model.safetensors'] + ltx2_ignore = [] + if selected.repo in ltx2_redundant_connector_repos or 'LTXVideo 2.5' in selected.name: + ltx2_ignore.append('connectors/diffusion_pytorch_model.safetensors') + if 'LTXVideo 2.5' in selected.name: + # the pipeline fetch pulls every model-index folder except passed components: transformer_full + # is not one, the diffusion decoder is a separate pipeline, the LoRA is fetched on demand + ltx2_ignore += ['transformer_full/*', 'diffusion_decoder/*', 'ltx-2.5-22b-distilled-lora-450-bf16.safetensors'] + if ltx2_ignore: + kwargs['ignore_patterns'] = ltx2_ignore # LTX2TextConnectors weights are byte-identical across all 2.3 variants (verified by blob # hash). Pre-load from a canonical repo so per-variant fetches skip connectors/ entirely. # FP16 variants share OzzyGT/LTX-2.3; SDNQ variants share the pre-quantized mirror. diff --git a/ui/locale/locale_en.json b/ui/locale/locale_en.json index 532f1c81c..b216e8c41 100644 --- a/ui/locale/locale_en.json +++ b/ui/locale/locale_en.json @@ -832,10 +832,11 @@ {"id":"","label":"LTX model","localized":"","hint":"","ui":"video"}, {"id":"","label":"LTX frames number","localized":"","hint":"","ui":"video"}, {"id":"","label":"LTX frames skip","localized":"","hint":"","ui":"video"}, - {"id":"","label":"LTX enable upsampling","localized":"","hint":"","ui":"video"}, - {"id":"","label":"LTX upsample ratio","localized":"","hint":"","ui":"video"}, - {"id":"","label":"LTX enable refine","localized":"","hint":"","ui":"video"}, - {"id":"","label":"LTX refine strength","localized":"","hint":"","ui":"video"}, + {"id":"","label":"LTX auto duration","localized":"","hint":"Clip length is predicted from the prompt and the frames setting is ignored","ui":"video"}, + {"id":"","label":"LTX upscale","localized":"","hint":"","ui":"video"}, + {"id":"","label":"LTX scale","localized":"","hint":"","ui":"video"}, + {"id":"","label":"LTX refine","localized":"","hint":"","ui":"video"}, + {"id":"","label":"LTX strength","localized":"","hint":"","ui":"video"}, {"id":"","label":"LTX decode timestep","localized":"","hint":"","ui":"video"}, {"id":"","label":"LTX save audio","localized":"","hint":"LTX-2 audio-capable models always generate audio from the same prompt as video; this toggle controls whether the audio track is included in the saved video file","ui":"video"}, {"id":"","label":"Loop","localized":"","hint":"","ui":"extras"}, From 5cb6efea340e9c63a4f1f2d5f18c712e3e4e4548 Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Thu, 13 Aug 2026 02:36:40 +0100 Subject: [PATCH 8/9] fix(ltx): apply the 2.5 stage-2 lora to the text connectors The stage 2 LoRA stores its connector deltas as diffusion_model.{video,audio}_embeddings_connector, but LTX2LoraLoaderMixin.lora_state_dict recognizes connectors only under the 2.3 text_embedding_projection prefix. All 3544 keys are routed into the transformer namespace and peft drops the 224 that land nowhere, leaving refine with a transformer-only adapter. Wrapping lora_state_dict moves those keys onto the connectors component using the rename table from the convert_ltx2_to_diffusers script. The wrapper is inert once no misrouted keys appear, so it needs no version check. --- modules/ltx/ltx_diffusers_patch.py | 103 ++++++++++++++++++++++------- 1 file changed, 80 insertions(+), 23 deletions(-) diff --git a/modules/ltx/ltx_diffusers_patch.py b/modules/ltx/ltx_diffusers_patch.py index 4444cb19f..43bef4ce3 100644 --- a/modules/ltx/ltx_diffusers_patch.py +++ b/modules/ltx/ltx_diffusers_patch.py @@ -1,29 +1,47 @@ -"""Workaround for huggingface/diffusers#13564 connectors padding regression. +"""Local fixes for LTX-2.x gaps in the pinned diffusers. -PR #13564 (merged 2026-05-08) refactored LTX2ConnectorTransformer1d's padding -logic from a loop-based gather-and-pad into a vectorized mask-then-flip. The -new code applies torch.flip(hidden_states, dims=[1]) after replacing padding -positions with learned registers, which reverses the order of valid prompt -tokens. Audio cross-attention is position-sensitive, so reversed token order -produces jumbled dialogue (right vocabulary, wrong word order). Visual quality -is mostly unaffected because spatial cross-attention is less position-sensitive. +Both patches are installed at import time by ltx_process and are safe to leave in +place once upstream fixes them: the first skips when the source no longer matches, +the second is a no-op as soon as no misrouted keys appear. -This module restores the pre-#13564 forward at import time when the broken -pattern is detected. Safe to leave in place after upstream fixes the bug: -detection will skip the monkey-patch when the source no longer matches. +Connector padding (huggingface/diffusers#13564): PR #13564 (merged 2026-05-08) +refactored LTX2ConnectorTransformer1d's padding logic from a loop-based +gather-and-pad into a vectorized mask-then-flip. The new code applies +torch.flip(hidden_states, dims=[1]) after replacing padding positions with learned +registers, which reverses the order of valid prompt tokens. Audio cross-attention is +position-sensitive, so reversed token order produces jumbled dialogue (right +vocabulary, wrong word order). Visual quality is mostly unaffected because spatial +cross-attention is less position-sensitive. + +Stage-2 LoRA connectors: LTX2LoraLoaderMixin.lora_state_dict recognizes connector +weights only under the 2.3-era text_embedding_projection prefix, so a +diffusion_model.* checkpoint is routed wholesale into the transformer namespace. The +2.5 stage-2 distilled LoRA carries its connector deltas as +diffusion_model.{video,audio}_embeddings_connector.*, so 224 of its 3544 keys reach a +module that cannot host them and peft drops them. Re-routing uses the rename table +from the convert_ltx2_to_diffusers script. """ +import functools import inspect import torch import torch.nn.functional as F - -_PATCH_APPLIED = False -_BROKEN_MARKER = 'torch.flip(hidden_states, dims=[1])' +from modules.logger import log -def _patched_forward( +PATCH_APPLIED = False +BROKEN_MARKER = 'torch.flip(hidden_states, dims=[1])' +CONNECTOR_LORA_PREFIXES = ('video_embeddings_connector.', 'audio_embeddings_connector.') +CONNECTOR_LORA_RENAME = { + 'video_embeddings_connector': 'video_connector', + 'audio_embeddings_connector': 'audio_connector', + 'transformer_1d_blocks': 'transformer_blocks', +} + + +def patched_connector_forward( self, hidden_states: torch.Tensor, attention_mask: torch.Tensor | None = None, @@ -72,19 +90,58 @@ def _patched_forward( return hidden_states, attention_mask -def apply_patch(): - global _PATCH_APPLIED # pylint: disable=global-statement - if _PATCH_APPLIED: - return +def reroute_connector_keys(state_dict): + converted = {} + moved = 0 + for key, value in state_dict.items(): + name = key.removeprefix('transformer.') + if name.startswith(CONNECTOR_LORA_PREFIXES): + for src, dst in CONNECTOR_LORA_RENAME.items(): + name = name.replace(src, dst) + converted[f'connectors.{name}'] = value + moved += 1 + else: + converted[key] = value + if moved == 0: + return state_dict + log.debug(f'LTX: lora=connectors rerouted={moved} total={len(state_dict)}') + return converted + + +def apply_connectors_forward_patch(): try: from diffusers.pipelines.ltx2.connectors import LTX2ConnectorTransformer1d except ImportError: - _PATCH_APPLIED = True return try: source = inspect.getsource(LTX2ConnectorTransformer1d.forward) except (OSError, TypeError): source = '' - if _BROKEN_MARKER in source: - LTX2ConnectorTransformer1d.forward = _patched_forward # TODO ltx: patched diffusers connectors padding to fix audio token order (upstream #13564 regression) - _PATCH_APPLIED = True + if BROKEN_MARKER in source: + LTX2ConnectorTransformer1d.forward = patched_connector_forward # TODO ltx: patched diffusers connectors padding to fix audio token order (upstream #13564 regression) + + +def apply_lora_patch(): + try: + from diffusers.loaders.lora_pipeline import LTX2LoraLoaderMixin + except ImportError: + return + original = LTX2LoraLoaderMixin.lora_state_dict.__func__ + + @functools.wraps(original) + def lora_state_dict(cls, *args, **kwargs): # TODO ltx: diffusers routes 2.5 stage-2 lora connector keys into the transformer namespace + loaded = original(cls, *args, **kwargs) + if isinstance(loaded, tuple): + return (reroute_connector_keys(loaded[0]), *loaded[1:]) + return reroute_connector_keys(loaded) + + LTX2LoraLoaderMixin.lora_state_dict = classmethod(lora_state_dict) + + +def apply_patch(): + global PATCH_APPLIED # pylint: disable=global-statement + if PATCH_APPLIED: + return + apply_connectors_forward_patch() + apply_lora_patch() + PATCH_APPLIED = True From f6607e89225eaa07d606e11ceee00b312998e13e Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Wed, 12 Aug 2026 02:24:41 +0100 Subject: [PATCH 9/9] fix(hub): install the token header hijack at startup Implicit tokens are disabled process-wide, so the hijack that re-attaches the token explicitly is what keeps gated repos reachable. It was installed by the checkpoint loader and the caption paths, so loading a video model into a fresh session sent unauthenticated requests and gated repos answered 401. --- modules/models_hf.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/modules/models_hf.py b/modules/models_hf.py index f6d311b8c..64f4f5b80 100644 --- a/modules/models_hf.py +++ b/modules/models_hf.py @@ -1,6 +1,7 @@ import os import time import gradio as gr +from modules import sd_hijack_hfhub from modules.logger import log from modules.shared import opts @@ -37,6 +38,10 @@ def hf_init(): obfuscated_token = 'hf_...' + opts.huggingface_token[-4:] log.info(f'Huggingface: transfer={opts.hf_transfer_mode} parallel={opts.sd_parallel_load} direct={opts.diffusers_to_gpu} token="{obfuscated_token}" cache="{opts.hfcache_dir}"') + # the disable flag above drops the token from token=None requests, which is what diffusers and + # transformers send; the hijack re-adds it explicitly and has to precede the first download + sd_hijack_hfhub.init_hijack() + def hf_check_cache(): t0 = time.time()