diff --git a/CHANGELOG.md b/CHANGELOG.md index 9df8c4e1e..df917041b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,8 @@ While we have several new supported models, workflows and tools, this release is with full search and tons of new documentation - New settings panel with simplified and streamlined configuration -We've also added support for several new models (see [supported models](https://vladmandic.github.io/sdnext-docs/Model-Support/) for full list) such as [NVLabs Sana](https://huggingface.co/Efficient-Large-Model/Sana_1600M_1024px) and [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video) +We've also added support for several new models (see [supported models](https://vladmandic.github.io/sdnext-docs/Model-Support/) for full list) such as highly anticipated [NVLabs Sana](https://huggingface.co/Efficient-Large-Model/Sana_1600M_1024px) +And several new video models: [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video), [Hunyuan Video](https://huggingface.co/tencent/HunyuanVideo) and [Genmo Mochi.1 Preview](https://huggingface.co/genmo/mochi-1-preview) And a lot of Control and IPAdapter goodies - for SDXL there is new [ProMax](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0), improved *Union* and *Tiling* @@ -70,11 +71,6 @@ And it wouldn't be a X-mass edition custom themes: *Snowflake* and *Elf-Green* both **Depth** and **Canny** LoRAs are available in standard control menus - [StabilityAI SD35 ControlNets](https://huggingface.co/stabilityai/stable-diffusion-3.5-controlnets) - In addition to previously released `InstantX` and `Alimama`, we now have *official* ones from StabilityAI -- [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video) - basic support for LTX-Video for text-to-video and image-to-video - to use, select in *scripts -> ltx-video* - *note* you may need to enable sequential offload for maximum gpu memory savings - *note* ltx-video requires very long and descriptive prompt, see original link for examples - [Style Aligned Image Generation](https://style-aligned-gen.github.io/) enable in scripts, compatible with sd-xl enter multiple prompts in prompt field separated by new line @@ -87,6 +83,30 @@ And it wouldn't be a X-mass edition custom themes: *Snowflake* and *Elf-Green* can render 4k sdxl images *note*: disable live preview to avoid memory issues when generating large images +### Video models + +- [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video) + model size: 27.75gb + support for text-to-video and image-to-video, to use, select in *scripts -> ltx-video* + *refrence values*: steps 50, width 704, height 512, frames 161, guidance scale 3.0 +- [Hunyuan Video](https://huggingface.co/tencent/HunyuanVideo) + model size: 40.92gb + support for text-to-video, to use, select in *scripts -> hunyuan video* + *refrence values*: steps 50, width 1280, height 720, frames 129, guidance scale 6.0 +- [Genmo Mochi.1 Preview](https://huggingface.co/genmo/mochi-1-preview) + support for text-to-video, to use, select in *scripts -> mochi.1 video* + *refrence values*: steps 64, width 848, height 480, frames 19, guidance scale 4.5 + +*Notes*: +- all video models are very large and resource intensive! + any use on gpus below 16gb and systems below 48gb ram is experimental at best +- sdnext support for video models is relatively basic with further optimizations pending community interest + any future optimizations would likely have to go into partial loading and excecution instead of offloading inactive parts of the model +- new video models use generic llms for prompting and due to that requires very long and descriptive prompt +- you may need to enable sequential offload for maximum gpu memory savings +- optionally enable pre-quantization using bnb for additional memory savings +- reduce number of frames and/or resolution to reduce memory usage + ### UI and workflow improvements - **Docs**: diff --git a/scripts/animatediff.py b/scripts/animatediff.py index 91db60915..6c29f3fa5 100644 --- a/scripts/animatediff.py +++ b/scripts/animatediff.py @@ -250,7 +250,7 @@ class Script(scripts.Script): processing.fix_seed(p) p.extra_generation_params['AnimateDiff'] = loaded_adapter p.do_not_save_grid = True - p.ops.append('animatediff') + p.ops.append('video') p.task_args['generator'] = None p.task_args['num_frames'] = frames p.task_args['num_inference_steps'] = p.steps diff --git a/scripts/cogvideo.py b/scripts/cogvideo.py index a5efcd3e6..e689a5e3f 100644 --- a/scripts/cogvideo.py +++ b/scripts/cogvideo.py @@ -202,7 +202,7 @@ class Script(scripts.Script): p.extra_generation_params['CogVideoX'] = model p.do_not_save_grid = True if 'animatediff' not in p.ops: - p.ops.append('cogvideox') + p.ops.append('video') if override: p.width = 720 p.height = 480 diff --git a/scripts/hunyuanvideo.py b/scripts/hunyuanvideo.py new file mode 100644 index 000000000..b94c8b8f8 --- /dev/null +++ b/scripts/hunyuanvideo.py @@ -0,0 +1,111 @@ +import time +import torch +import gradio as gr +import diffusers +from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint, model_quant + + +repo_id = 'tencent/HunyuanVideo' +""" +prompt_template = { # default + "template": ( + "<|start_header_id|>system<|end_header_id|>\n\nDescribe the video by detailing the following aspects: " + "1. The main content and theme of the video." + "2. The color, shape, size, texture, quantity, text, and spatial relationships of the contents, including objects, people, and anything else." + "3. Actions, events, behaviors temporal relationships, physical movement changes of the contents." + "4. Background environment, light, style, atmosphere, and qualities." + "5. Camera angles, movements, and transitions used in the video." + "6. Thematic and aesthetic concepts associated with the scene, i.e. realistic, futuristic, fairy tale, etc<|eot_id|>" + "<|start_header_id|>user<|end_header_id|>\n\n{}<|eot_id|>" + ), + "crop_start": 95, +} +""" + + +class Script(scripts.Script): + def title(self): + return 'Video: Hunyuan Video' + + def show(self, is_img2img): + return not is_img2img if shared.native else False + + # return signature is array of gradio components + def ui(self, _is_img2img): + def video_type_change(video_type): + return [ + gr.update(visible=video_type != 'None'), + gr.update(visible=video_type == 'GIF' or video_type == 'PNG'), + gr.update(visible=video_type == 'MP4'), + gr.update(visible=video_type == 'MP4'), + ] + + with gr.Row(): + gr.HTML('  Hunyuan Video
') + with gr.Row(): + num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=45) + with gr.Row(): + video_type = gr.Dropdown(label='Video file', choices=['None', 'GIF', 'PNG', 'MP4'], value='None') + duration = gr.Slider(label='Duration', minimum=0.25, maximum=10, step=0.25, value=2, visible=False) + with gr.Row(): + gif_loop = gr.Checkbox(label='Loop', value=True, visible=False) + mp4_pad = gr.Slider(label='Pad frames', minimum=0, maximum=24, step=1, value=1, visible=False) + mp4_interpolate = gr.Slider(label='Interpolate frames', minimum=0, maximum=24, step=1, value=0, visible=False) + video_type.change(fn=video_type_change, inputs=[video_type], outputs=[duration, gif_loop, mp4_pad, mp4_interpolate]) + return [num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate] + + def run(self, p: processing.StableDiffusionProcessing, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument + # set params + num_frames = int(num_frames) + p.width = 32 * int(p.width // 32) + p.height = 32 * int(p.height // 32) + p.task_args['output_type'] = 'pil' + p.task_args['generator'] = torch.manual_seed(p.seed) + p.task_args['num_frames'] = num_frames + # p.task_args['prompt_template'] = prompt_template + p.sampler_name = 'Default' + p.do_not_save_grid = True + p.ops.append('video') + + # load model + cls = diffusers.HunyuanVideoPipeline + if shared.sd_model.__class__ != cls: + sd_models.unload_model_weights() + kwargs = {} + kwargs = model_quant.create_bnb_config(kwargs) + kwargs = model_quant.create_ao_config(kwargs) + transformer = diffusers.HunyuanVideoTransformer3DModel.from_pretrained( + repo_id, + subfolder="transformer", + torch_dtype=devices.dtype, + revision="refs/pr/18", + cache_dir = shared.opts.hfcache_dir, + **kwargs + ) + shared.sd_model = cls.from_pretrained( + repo_id, + transformer=transformer, + revision="refs/pr/18", + cache_dir = shared.opts.hfcache_dir, + torch_dtype=devices.dtype, + **kwargs + ) + shared.sd_model.scheduler._shift = 7.0 # pylint: disable=protected-access + sd_models.set_diffuser_options(shared.sd_model) + shared.sd_model.sd_checkpoint_info = sd_checkpoint.CheckpointInfo(repo_id) + shared.sd_model.sd_model_hash = None + shared.sd_model = sd_models.apply_balanced_offload(shared.sd_model) + shared.sd_model.vae.enable_slicing() + shared.sd_model.vae.enable_tiling() + devices.torch_gc(force=True) + shared.log.debug(f'Video: cls={shared.sd_model.__class__.__name__} args={p.task_args}') + + # run processing + t0 = time.time() + processed = processing.process_images(p) + t1 = time.time() + if processed is not None and len(processed.images) > 0: + shared.log.info(f'Video: frames={len(processed.images)} time={t1-t0:.2f}') + if video_type != 'None': + images.save_video(p, filename=None, images=processed.images, video_type=video_type, duration=duration, loop=gif_loop, pad=mp4_pad, interpolate=mp4_interpolate) + return processed diff --git a/scripts/image2video.py b/scripts/image2video.py index 5e08922ee..ad6615f67 100644 --- a/scripts/image2video.py +++ b/scripts/image2video.py @@ -73,7 +73,7 @@ class Script(scripts.Script): model = [m for m in MODELS if m['name'] == model_name][0] repo_id = model['url'] shared.log.debug(f'Image2Video: model={model_name} frames={num_frames}, video={video_type} duration={duration} loop={gif_loop} pad={mp4_pad} interpolate={mp4_interpolate}') - p.ops.append('image2video') + p.ops.append('video') p.do_not_save_grid = True orig_pipeline = shared.sd_model diff --git a/scripts/ltxvideo.py b/scripts/ltxvideo.py index 54e2685a8..50530563a 100644 --- a/scripts/ltxvideo.py +++ b/scripts/ltxvideo.py @@ -2,40 +2,10 @@ import time import torch import gradio as gr import diffusers -from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint +from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint, model_quant repo_id = 'a-r-r-o-w/LTX-Video-diffusers' -presets = [ - {"label": "custom", "width": 0, "height": 0, "num_frames": 0}, - {"label": "1216x704, 41 frames", "width": 1216, "height": 704, "num_frames": 41}, - {"label": "1088x704, 49 frames", "width": 1088, "height": 704, "num_frames": 49}, - {"label": "1056x640, 57 frames", "width": 1056, "height": 640, "num_frames": 57}, - {"label": "992x608, 65 frames", "width": 992, "height": 608, "num_frames": 65}, - {"label": "896x608, 73 frames", "width": 896, "height": 608, "num_frames": 73}, - {"label": "896x544, 81 frames", "width": 896, "height": 544, "num_frames": 81}, - {"label": "832x544, 89 frames", "width": 832, "height": 544, "num_frames": 89}, - {"label": "800x512, 97 frames", "width": 800, "height": 512, "num_frames": 97}, - {"label": "768x512, 97 frames", "width": 768, "height": 512, "num_frames": 97}, - {"label": "800x480, 105 frames", "width": 800, "height": 480, "num_frames": 105}, - {"label": "736x480, 113 frames", "width": 736, "height": 480, "num_frames": 113}, - {"label": "704x480, 121 frames", "width": 704, "height": 480, "num_frames": 121}, - {"label": "704x448, 129 frames", "width": 704, "height": 448, "num_frames": 129}, - {"label": "672x448, 137 frames", "width": 672, "height": 448, "num_frames": 137}, - {"label": "640x416, 153 frames", "width": 640, "height": 416, "num_frames": 153}, - {"label": "672x384, 161 frames", "width": 672, "height": 384, "num_frames": 161}, - {"label": "640x384, 169 frames", "width": 640, "height": 384, "num_frames": 169}, - {"label": "608x384, 177 frames", "width": 608, "height": 384, "num_frames": 177}, - {"label": "576x384, 185 frames", "width": 576, "height": 384, "num_frames": 185}, - {"label": "608x352, 193 frames", "width": 608, "height": 352, "num_frames": 193}, - {"label": "576x352, 201 frames", "width": 576, "height": 352, "num_frames": 201}, - {"label": "544x352, 209 frames", "width": 544, "height": 352, "num_frames": 209}, - {"label": "512x352, 225 frames", "width": 512, "height": 352, "num_frames": 225}, - {"label": "512x352, 233 frames", "width": 512, "height": 352, "num_frames": 233}, - {"label": "544x320, 241 frames", "width": 544, "height": 320, "num_frames": 241}, - {"label": "512x320, 249 frames", "width": 512, "height": 320, "num_frames": 249}, - {"label": "512x320, 257 frames", "width": 512, "height": 320, "num_frames": 257}, -] class Script(scripts.Script): @@ -54,14 +24,11 @@ class Script(scripts.Script): gr.update(visible=video_type == 'MP4'), gr.update(visible=video_type == 'MP4'), ] - def preset_change(preset): - return gr.update(visible=preset == 'custom') with gr.Row(): gr.HTML('  LTX Video
') with gr.Row(): - preset_name = gr.Dropdown(label='Preset', choices=[p['label'] for p in presets], value='custom') - num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=9) + num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=41) with gr.Row(): video_type = gr.Dropdown(label='Video file', choices=['None', 'GIF', 'PNG', 'MP4'], value='None') duration = gr.Slider(label='Duration', minimum=0.25, maximum=10, step=0.25, value=2, visible=False) @@ -69,26 +36,19 @@ class Script(scripts.Script): gif_loop = gr.Checkbox(label='Loop', value=True, visible=False) mp4_pad = gr.Slider(label='Pad frames', minimum=0, maximum=24, step=1, value=1, visible=False) mp4_interpolate = gr.Slider(label='Interpolate frames', minimum=0, maximum=24, step=1, value=0, visible=False) - preset_name.change(fn=preset_change, inputs=[preset_name], outputs=num_frames) video_type.change(fn=video_type_change, inputs=[video_type], outputs=[duration, gif_loop, mp4_pad, mp4_interpolate]) - return [preset_name, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate] + return [num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate] - def run(self, p: processing.StableDiffusionProcessing, preset_name, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument + def run(self, p: processing.StableDiffusionProcessing, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument # set params - preset = [p for p in presets if p['label'] == preset_name][0] image = getattr(p, 'init_images', None) image = None if image is None or len(image) == 0 else image[0] if p.width == 0 or p.height == 0 and image is not None: p.width = image.width p.height = image.height - if preset['label'] != 'custom': - num_frames = preset['num_frames'] - p.width = preset['width'] - p.height = preset['height'] - else: - num_frames = 8 * int(num_frames // 8) + 1 - p.width = 32 * int(p.width // 32) - p.height = 32 * int(p.height // 32) + num_frames = 8 * int(num_frames // 8) + 1 + p.width = 32 * int(p.width // 32) + p.height = 32 * int(p.height // 32) if image: image = images.resize_image(resize_mode=2, im=image, width=p.width, height=p.height, upscaler_name=None, output_type='pil') p.task_args['image'] = image @@ -97,7 +57,7 @@ class Script(scripts.Script): p.task_args['num_frames'] = num_frames p.sampler_name = 'Default' p.do_not_save_grid = True - p.ops.append('ltx') + p.ops.append('video') # load model cls = diffusers.LTXPipeline if image is None else diffusers.LTXImageToVideoPipeline @@ -105,10 +65,14 @@ class Script(scripts.Script): diffusers.AutoencoderKLLTX = diffusers.AutoencoderKLLTXVideo if shared.sd_model.__class__ != cls: sd_models.unload_model_weights() + kwargs = {} + kwargs = model_quant.create_bnb_config(kwargs) + kwargs = model_quant.create_ao_config(kwargs) shared.sd_model = cls.from_pretrained( repo_id, cache_dir = shared.opts.hfcache_dir, torch_dtype=devices.dtype, + **kwargs ) sd_models.set_diffuser_options(shared.sd_model) shared.sd_model.sd_checkpoint_info = sd_checkpoint.CheckpointInfo(repo_id) @@ -117,14 +81,14 @@ class Script(scripts.Script): shared.sd_model.vae.enable_slicing() shared.sd_model.vae.enable_tiling() devices.torch_gc(force=True) - shared.log.debug(f'LTX: cls={shared.sd_model.__class__.__name__} preset={preset_name} args={p.task_args}') + shared.log.debug(f'Video: cls={shared.sd_model.__class__.__name__} args={p.task_args}') # run processing t0 = time.time() processed = processing.process_images(p) t1 = time.time() if processed is not None and len(processed.images) > 0: - shared.log.info(f'LTX: frames={len(processed.images)} time={t1-t0:.2f}') + shared.log.info(f'Video: frames={len(processed.images)} time={t1-t0:.2f}') if video_type != 'None': images.save_video(p, filename=None, images=processed.images, video_type=video_type, duration=duration, loop=gif_loop, pad=mp4_pad, interpolate=mp4_interpolate) return processed diff --git a/scripts/mochivideo.py b/scripts/mochivideo.py new file mode 100644 index 000000000..f85616a5e --- /dev/null +++ b/scripts/mochivideo.py @@ -0,0 +1,85 @@ +import time +import torch +import gradio as gr +import diffusers +from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint, model_quant + + +repo_id = 'genmo/mochi-1-preview' + + +class Script(scripts.Script): + def title(self): + return 'Video: Mochi.1 Video' + + def show(self, is_img2img): + return not is_img2img if shared.native else False + + # return signature is array of gradio components + def ui(self, _is_img2img): + def video_type_change(video_type): + return [ + gr.update(visible=video_type != 'None'), + gr.update(visible=video_type == 'GIF' or video_type == 'PNG'), + gr.update(visible=video_type == 'MP4'), + gr.update(visible=video_type == 'MP4'), + ] + + with gr.Row(): + gr.HTML('  Mochi.1 Video
') + with gr.Row(): + num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=45) + with gr.Row(): + video_type = gr.Dropdown(label='Video file', choices=['None', 'GIF', 'PNG', 'MP4'], value='None') + duration = gr.Slider(label='Duration', minimum=0.25, maximum=10, step=0.25, value=2, visible=False) + with gr.Row(): + gif_loop = gr.Checkbox(label='Loop', value=True, visible=False) + mp4_pad = gr.Slider(label='Pad frames', minimum=0, maximum=24, step=1, value=1, visible=False) + mp4_interpolate = gr.Slider(label='Interpolate frames', minimum=0, maximum=24, step=1, value=0, visible=False) + video_type.change(fn=video_type_change, inputs=[video_type], outputs=[duration, gif_loop, mp4_pad, mp4_interpolate]) + return [num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate] + + def run(self, p: processing.StableDiffusionProcessing, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument + # set params + num_frames = int(num_frames // 8) + p.width = 32 * int(p.width // 32) + p.height = 32 * int(p.height // 32) + p.task_args['output_type'] = 'pil' + p.task_args['generator'] = torch.manual_seed(p.seed) + p.task_args['num_frames'] = num_frames + p.sampler_name = 'Default' + p.do_not_save_grid = True + p.ops.append('video') + + # load model + cls = diffusers.MochiPipeline + if shared.sd_model.__class__ != cls: + sd_models.unload_model_weights() + kwargs = {} + kwargs = model_quant.create_bnb_config(kwargs) + kwargs = model_quant.create_ao_config(kwargs) + shared.sd_model = cls.from_pretrained( + repo_id, + cache_dir = shared.opts.hfcache_dir, + torch_dtype=devices.dtype, + **kwargs + ) + shared.sd_model.scheduler._shift = 7.0 # pylint: disable=protected-access + sd_models.set_diffuser_options(shared.sd_model) + shared.sd_model.sd_checkpoint_info = sd_checkpoint.CheckpointInfo(repo_id) + shared.sd_model.sd_model_hash = None + shared.sd_model = sd_models.apply_balanced_offload(shared.sd_model) + shared.sd_model.vae.enable_slicing() + shared.sd_model.vae.enable_tiling() + devices.torch_gc(force=True) + shared.log.debug(f'Video: cls={shared.sd_model.__class__.__name__} args={p.task_args}') + + # run processing + t0 = time.time() + processed = processing.process_images(p) + t1 = time.time() + if processed is not None and len(processed.images) > 0: + shared.log.info(f'Video: frames={len(processed.images)} time={t1-t0:.2f}') + if video_type != 'None': + images.save_video(p, filename=None, images=processed.images, video_type=video_type, duration=duration, loop=gif_loop, pad=mp4_pad, interpolate=mp4_interpolate) + return processed diff --git a/scripts/stablevideodiffusion.py b/scripts/stablevideodiffusion.py index cbf2ce003..f8da35b23 100644 --- a/scripts/stablevideodiffusion.py +++ b/scripts/stablevideodiffusion.py @@ -81,7 +81,7 @@ class Script(scripts.Script): p.width = 1024 p.height = 576 image = images.resize_image(resize_mode=2, im=image, width=p.width, height=p.height, upscaler_name=None, output_type='pil') - p.ops.append('svd') + p.ops.append('video') p.do_not_save_grid = True p.init_images = [image] p.sampler_name = 'Default' # svd does not support non-default sampler diff --git a/scripts/text2video.py b/scripts/text2video.py index dc4c44cac..c7b3d1c05 100644 --- a/scripts/text2video.py +++ b/scripts/text2video.py @@ -87,7 +87,7 @@ class Script(scripts.Script): shared.opts.sd_model_checkpoint = checkpoint.name sd_models.reload_model_weights(op='model') - p.ops.append('text2video') + p.ops.append('video') p.do_not_save_grid = True if use_default: p.task_args['num_frames'] = model['params'][0] diff --git a/wiki b/wiki index 34ba1df45..a6c10ce38 160000 --- a/wiki +++ b/wiki @@ -1 +1 @@ -Subproject commit 34ba1df45d17da4ee09a2e5278e384bc1929dd8b +Subproject commit a6c10ce38ef1da4d47cd68ad0aa8552d2d62c943