add hunyuan video and mochi video

Signed-off-by: Vladimir Mandic <mandic00@live.com>
This commit is contained in:
Vladimir Mandic
2024-12-19 14:52:29 -05:00
parent b569977895
commit 15ea72ed74
10 changed files with 242 additions and 62 deletions
+26 -6
View File
@@ -14,7 +14,8 @@ While we have several new supported models, workflows and tools, this release is
with full search and tons of new documentation
- New settings panel with simplified and streamlined configuration
We've also added support for several new models (see [supported models](https://vladmandic.github.io/sdnext-docs/Model-Support/) for full list) such as [NVLabs Sana](https://huggingface.co/Efficient-Large-Model/Sana_1600M_1024px) and [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video)
We've also added support for several new models (see [supported models](https://vladmandic.github.io/sdnext-docs/Model-Support/) for full list) such as highly anticipated [NVLabs Sana](https://huggingface.co/Efficient-Large-Model/Sana_1600M_1024px)
And several new video models: [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video), [Hunyuan Video](https://huggingface.co/tencent/HunyuanVideo) and [Genmo Mochi.1 Preview](https://huggingface.co/genmo/mochi-1-preview)
And a lot of Control and IPAdapter goodies
- for SDXL there is new [ProMax](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0), improved *Union* and *Tiling*
@@ -70,11 +71,6 @@ And it wouldn't be a X-mass edition custom themes: *Snowflake* and *Elf-Green*
both **Depth** and **Canny** LoRAs are available in standard control menus
- [StabilityAI SD35 ControlNets](https://huggingface.co/stabilityai/stable-diffusion-3.5-controlnets)
- In addition to previously released `InstantX` and `Alimama`, we now have *official* ones from StabilityAI
- [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video)
basic support for LTX-Video for text-to-video and image-to-video
to use, select in *scripts -> ltx-video*
*note* you may need to enable sequential offload for maximum gpu memory savings
*note* ltx-video requires very long and descriptive prompt, see original link for examples
- [Style Aligned Image Generation](https://style-aligned-gen.github.io/)
enable in scripts, compatible with sd-xl
enter multiple prompts in prompt field separated by new line
@@ -87,6 +83,30 @@ And it wouldn't be a X-mass edition custom themes: *Snowflake* and *Elf-Green*
can render 4k sdxl images
*note*: disable live preview to avoid memory issues when generating large images
### Video models
- [Lightricks LTX-Video](https://huggingface.co/Lightricks/LTX-Video)
model size: 27.75gb
support for text-to-video and image-to-video, to use, select in *scripts -> ltx-video*
*refrence values*: steps 50, width 704, height 512, frames 161, guidance scale 3.0
- [Hunyuan Video](https://huggingface.co/tencent/HunyuanVideo)
model size: 40.92gb
support for text-to-video, to use, select in *scripts -> hunyuan video*
*refrence values*: steps 50, width 1280, height 720, frames 129, guidance scale 6.0
- [Genmo Mochi.1 Preview](https://huggingface.co/genmo/mochi-1-preview)
support for text-to-video, to use, select in *scripts -> mochi.1 video*
*refrence values*: steps 64, width 848, height 480, frames 19, guidance scale 4.5
*Notes*:
- all video models are very large and resource intensive!
any use on gpus below 16gb and systems below 48gb ram is experimental at best
- sdnext support for video models is relatively basic with further optimizations pending community interest
any future optimizations would likely have to go into partial loading and excecution instead of offloading inactive parts of the model
- new video models use generic llms for prompting and due to that requires very long and descriptive prompt
- you may need to enable sequential offload for maximum gpu memory savings
- optionally enable pre-quantization using bnb for additional memory savings
- reduce number of frames and/or resolution to reduce memory usage
### UI and workflow improvements
- **Docs**:
+1 -1
View File
@@ -250,7 +250,7 @@ class Script(scripts.Script):
processing.fix_seed(p)
p.extra_generation_params['AnimateDiff'] = loaded_adapter
p.do_not_save_grid = True
p.ops.append('animatediff')
p.ops.append('video')
p.task_args['generator'] = None
p.task_args['num_frames'] = frames
p.task_args['num_inference_steps'] = p.steps
+1 -1
View File
@@ -202,7 +202,7 @@ class Script(scripts.Script):
p.extra_generation_params['CogVideoX'] = model
p.do_not_save_grid = True
if 'animatediff' not in p.ops:
p.ops.append('cogvideox')
p.ops.append('video')
if override:
p.width = 720
p.height = 480
+111
View File
@@ -0,0 +1,111 @@
import time
import torch
import gradio as gr
import diffusers
from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint, model_quant
repo_id = 'tencent/HunyuanVideo'
"""
prompt_template = { # default
"template": (
"<|start_header_id|>system<|end_header_id|>\n\nDescribe the video by detailing the following aspects: "
"1. The main content and theme of the video."
"2. The color, shape, size, texture, quantity, text, and spatial relationships of the contents, including objects, people, and anything else."
"3. Actions, events, behaviors temporal relationships, physical movement changes of the contents."
"4. Background environment, light, style, atmosphere, and qualities."
"5. Camera angles, movements, and transitions used in the video."
"6. Thematic and aesthetic concepts associated with the scene, i.e. realistic, futuristic, fairy tale, etc<|eot_id|>"
"<|start_header_id|>user<|end_header_id|>\n\n{}<|eot_id|>"
),
"crop_start": 95,
}
"""
class Script(scripts.Script):
def title(self):
return 'Video: Hunyuan Video'
def show(self, is_img2img):
return not is_img2img if shared.native else False
# return signature is array of gradio components
def ui(self, _is_img2img):
def video_type_change(video_type):
return [
gr.update(visible=video_type != 'None'),
gr.update(visible=video_type == 'GIF' or video_type == 'PNG'),
gr.update(visible=video_type == 'MP4'),
gr.update(visible=video_type == 'MP4'),
]
with gr.Row():
gr.HTML('<a href="https://huggingface.co/tencent/HunyuanVideo">&nbsp Hunyuan Video</a><br>')
with gr.Row():
num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=45)
with gr.Row():
video_type = gr.Dropdown(label='Video file', choices=['None', 'GIF', 'PNG', 'MP4'], value='None')
duration = gr.Slider(label='Duration', minimum=0.25, maximum=10, step=0.25, value=2, visible=False)
with gr.Row():
gif_loop = gr.Checkbox(label='Loop', value=True, visible=False)
mp4_pad = gr.Slider(label='Pad frames', minimum=0, maximum=24, step=1, value=1, visible=False)
mp4_interpolate = gr.Slider(label='Interpolate frames', minimum=0, maximum=24, step=1, value=0, visible=False)
video_type.change(fn=video_type_change, inputs=[video_type], outputs=[duration, gif_loop, mp4_pad, mp4_interpolate])
return [num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate]
def run(self, p: processing.StableDiffusionProcessing, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument
# set params
num_frames = int(num_frames)
p.width = 32 * int(p.width // 32)
p.height = 32 * int(p.height // 32)
p.task_args['output_type'] = 'pil'
p.task_args['generator'] = torch.manual_seed(p.seed)
p.task_args['num_frames'] = num_frames
# p.task_args['prompt_template'] = prompt_template
p.sampler_name = 'Default'
p.do_not_save_grid = True
p.ops.append('video')
# load model
cls = diffusers.HunyuanVideoPipeline
if shared.sd_model.__class__ != cls:
sd_models.unload_model_weights()
kwargs = {}
kwargs = model_quant.create_bnb_config(kwargs)
kwargs = model_quant.create_ao_config(kwargs)
transformer = diffusers.HunyuanVideoTransformer3DModel.from_pretrained(
repo_id,
subfolder="transformer",
torch_dtype=devices.dtype,
revision="refs/pr/18",
cache_dir = shared.opts.hfcache_dir,
**kwargs
)
shared.sd_model = cls.from_pretrained(
repo_id,
transformer=transformer,
revision="refs/pr/18",
cache_dir = shared.opts.hfcache_dir,
torch_dtype=devices.dtype,
**kwargs
)
shared.sd_model.scheduler._shift = 7.0 # pylint: disable=protected-access
sd_models.set_diffuser_options(shared.sd_model)
shared.sd_model.sd_checkpoint_info = sd_checkpoint.CheckpointInfo(repo_id)
shared.sd_model.sd_model_hash = None
shared.sd_model = sd_models.apply_balanced_offload(shared.sd_model)
shared.sd_model.vae.enable_slicing()
shared.sd_model.vae.enable_tiling()
devices.torch_gc(force=True)
shared.log.debug(f'Video: cls={shared.sd_model.__class__.__name__} args={p.task_args}')
# run processing
t0 = time.time()
processed = processing.process_images(p)
t1 = time.time()
if processed is not None and len(processed.images) > 0:
shared.log.info(f'Video: frames={len(processed.images)} time={t1-t0:.2f}')
if video_type != 'None':
images.save_video(p, filename=None, images=processed.images, video_type=video_type, duration=duration, loop=gif_loop, pad=mp4_pad, interpolate=mp4_interpolate)
return processed
+1 -1
View File
@@ -73,7 +73,7 @@ class Script(scripts.Script):
model = [m for m in MODELS if m['name'] == model_name][0]
repo_id = model['url']
shared.log.debug(f'Image2Video: model={model_name} frames={num_frames}, video={video_type} duration={duration} loop={gif_loop} pad={mp4_pad} interpolate={mp4_interpolate}')
p.ops.append('image2video')
p.ops.append('video')
p.do_not_save_grid = True
orig_pipeline = shared.sd_model
+14 -50
View File
@@ -2,40 +2,10 @@ import time
import torch
import gradio as gr
import diffusers
from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint
from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint, model_quant
repo_id = 'a-r-r-o-w/LTX-Video-diffusers'
presets = [
{"label": "custom", "width": 0, "height": 0, "num_frames": 0},
{"label": "1216x704, 41 frames", "width": 1216, "height": 704, "num_frames": 41},
{"label": "1088x704, 49 frames", "width": 1088, "height": 704, "num_frames": 49},
{"label": "1056x640, 57 frames", "width": 1056, "height": 640, "num_frames": 57},
{"label": "992x608, 65 frames", "width": 992, "height": 608, "num_frames": 65},
{"label": "896x608, 73 frames", "width": 896, "height": 608, "num_frames": 73},
{"label": "896x544, 81 frames", "width": 896, "height": 544, "num_frames": 81},
{"label": "832x544, 89 frames", "width": 832, "height": 544, "num_frames": 89},
{"label": "800x512, 97 frames", "width": 800, "height": 512, "num_frames": 97},
{"label": "768x512, 97 frames", "width": 768, "height": 512, "num_frames": 97},
{"label": "800x480, 105 frames", "width": 800, "height": 480, "num_frames": 105},
{"label": "736x480, 113 frames", "width": 736, "height": 480, "num_frames": 113},
{"label": "704x480, 121 frames", "width": 704, "height": 480, "num_frames": 121},
{"label": "704x448, 129 frames", "width": 704, "height": 448, "num_frames": 129},
{"label": "672x448, 137 frames", "width": 672, "height": 448, "num_frames": 137},
{"label": "640x416, 153 frames", "width": 640, "height": 416, "num_frames": 153},
{"label": "672x384, 161 frames", "width": 672, "height": 384, "num_frames": 161},
{"label": "640x384, 169 frames", "width": 640, "height": 384, "num_frames": 169},
{"label": "608x384, 177 frames", "width": 608, "height": 384, "num_frames": 177},
{"label": "576x384, 185 frames", "width": 576, "height": 384, "num_frames": 185},
{"label": "608x352, 193 frames", "width": 608, "height": 352, "num_frames": 193},
{"label": "576x352, 201 frames", "width": 576, "height": 352, "num_frames": 201},
{"label": "544x352, 209 frames", "width": 544, "height": 352, "num_frames": 209},
{"label": "512x352, 225 frames", "width": 512, "height": 352, "num_frames": 225},
{"label": "512x352, 233 frames", "width": 512, "height": 352, "num_frames": 233},
{"label": "544x320, 241 frames", "width": 544, "height": 320, "num_frames": 241},
{"label": "512x320, 249 frames", "width": 512, "height": 320, "num_frames": 249},
{"label": "512x320, 257 frames", "width": 512, "height": 320, "num_frames": 257},
]
class Script(scripts.Script):
@@ -54,14 +24,11 @@ class Script(scripts.Script):
gr.update(visible=video_type == 'MP4'),
gr.update(visible=video_type == 'MP4'),
]
def preset_change(preset):
return gr.update(visible=preset == 'custom')
with gr.Row():
gr.HTML('<a href="https://www.ltxvideo.org/">&nbsp LTX Video</a><br>')
with gr.Row():
preset_name = gr.Dropdown(label='Preset', choices=[p['label'] for p in presets], value='custom')
num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=9)
num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=41)
with gr.Row():
video_type = gr.Dropdown(label='Video file', choices=['None', 'GIF', 'PNG', 'MP4'], value='None')
duration = gr.Slider(label='Duration', minimum=0.25, maximum=10, step=0.25, value=2, visible=False)
@@ -69,26 +36,19 @@ class Script(scripts.Script):
gif_loop = gr.Checkbox(label='Loop', value=True, visible=False)
mp4_pad = gr.Slider(label='Pad frames', minimum=0, maximum=24, step=1, value=1, visible=False)
mp4_interpolate = gr.Slider(label='Interpolate frames', minimum=0, maximum=24, step=1, value=0, visible=False)
preset_name.change(fn=preset_change, inputs=[preset_name], outputs=num_frames)
video_type.change(fn=video_type_change, inputs=[video_type], outputs=[duration, gif_loop, mp4_pad, mp4_interpolate])
return [preset_name, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate]
return [num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate]
def run(self, p: processing.StableDiffusionProcessing, preset_name, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument
def run(self, p: processing.StableDiffusionProcessing, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument
# set params
preset = [p for p in presets if p['label'] == preset_name][0]
image = getattr(p, 'init_images', None)
image = None if image is None or len(image) == 0 else image[0]
if p.width == 0 or p.height == 0 and image is not None:
p.width = image.width
p.height = image.height
if preset['label'] != 'custom':
num_frames = preset['num_frames']
p.width = preset['width']
p.height = preset['height']
else:
num_frames = 8 * int(num_frames // 8) + 1
p.width = 32 * int(p.width // 32)
p.height = 32 * int(p.height // 32)
num_frames = 8 * int(num_frames // 8) + 1
p.width = 32 * int(p.width // 32)
p.height = 32 * int(p.height // 32)
if image:
image = images.resize_image(resize_mode=2, im=image, width=p.width, height=p.height, upscaler_name=None, output_type='pil')
p.task_args['image'] = image
@@ -97,7 +57,7 @@ class Script(scripts.Script):
p.task_args['num_frames'] = num_frames
p.sampler_name = 'Default'
p.do_not_save_grid = True
p.ops.append('ltx')
p.ops.append('video')
# load model
cls = diffusers.LTXPipeline if image is None else diffusers.LTXImageToVideoPipeline
@@ -105,10 +65,14 @@ class Script(scripts.Script):
diffusers.AutoencoderKLLTX = diffusers.AutoencoderKLLTXVideo
if shared.sd_model.__class__ != cls:
sd_models.unload_model_weights()
kwargs = {}
kwargs = model_quant.create_bnb_config(kwargs)
kwargs = model_quant.create_ao_config(kwargs)
shared.sd_model = cls.from_pretrained(
repo_id,
cache_dir = shared.opts.hfcache_dir,
torch_dtype=devices.dtype,
**kwargs
)
sd_models.set_diffuser_options(shared.sd_model)
shared.sd_model.sd_checkpoint_info = sd_checkpoint.CheckpointInfo(repo_id)
@@ -117,14 +81,14 @@ class Script(scripts.Script):
shared.sd_model.vae.enable_slicing()
shared.sd_model.vae.enable_tiling()
devices.torch_gc(force=True)
shared.log.debug(f'LTX: cls={shared.sd_model.__class__.__name__} preset={preset_name} args={p.task_args}')
shared.log.debug(f'Video: cls={shared.sd_model.__class__.__name__} args={p.task_args}')
# run processing
t0 = time.time()
processed = processing.process_images(p)
t1 = time.time()
if processed is not None and len(processed.images) > 0:
shared.log.info(f'LTX: frames={len(processed.images)} time={t1-t0:.2f}')
shared.log.info(f'Video: frames={len(processed.images)} time={t1-t0:.2f}')
if video_type != 'None':
images.save_video(p, filename=None, images=processed.images, video_type=video_type, duration=duration, loop=gif_loop, pad=mp4_pad, interpolate=mp4_interpolate)
return processed
+85
View File
@@ -0,0 +1,85 @@
import time
import torch
import gradio as gr
import diffusers
from modules import scripts, processing, shared, images, devices, sd_models, sd_checkpoint, model_quant
repo_id = 'genmo/mochi-1-preview'
class Script(scripts.Script):
def title(self):
return 'Video: Mochi.1 Video'
def show(self, is_img2img):
return not is_img2img if shared.native else False
# return signature is array of gradio components
def ui(self, _is_img2img):
def video_type_change(video_type):
return [
gr.update(visible=video_type != 'None'),
gr.update(visible=video_type == 'GIF' or video_type == 'PNG'),
gr.update(visible=video_type == 'MP4'),
gr.update(visible=video_type == 'MP4'),
]
with gr.Row():
gr.HTML('<a href="https://huggingface.co/genmo/mochi-1-preview">&nbsp Mochi.1 Video</a><br>')
with gr.Row():
num_frames = gr.Slider(label='Frames', minimum=9, maximum=257, step=1, value=45)
with gr.Row():
video_type = gr.Dropdown(label='Video file', choices=['None', 'GIF', 'PNG', 'MP4'], value='None')
duration = gr.Slider(label='Duration', minimum=0.25, maximum=10, step=0.25, value=2, visible=False)
with gr.Row():
gif_loop = gr.Checkbox(label='Loop', value=True, visible=False)
mp4_pad = gr.Slider(label='Pad frames', minimum=0, maximum=24, step=1, value=1, visible=False)
mp4_interpolate = gr.Slider(label='Interpolate frames', minimum=0, maximum=24, step=1, value=0, visible=False)
video_type.change(fn=video_type_change, inputs=[video_type], outputs=[duration, gif_loop, mp4_pad, mp4_interpolate])
return [num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate]
def run(self, p: processing.StableDiffusionProcessing, num_frames, video_type, duration, gif_loop, mp4_pad, mp4_interpolate): # pylint: disable=arguments-differ, unused-argument
# set params
num_frames = int(num_frames // 8)
p.width = 32 * int(p.width // 32)
p.height = 32 * int(p.height // 32)
p.task_args['output_type'] = 'pil'
p.task_args['generator'] = torch.manual_seed(p.seed)
p.task_args['num_frames'] = num_frames
p.sampler_name = 'Default'
p.do_not_save_grid = True
p.ops.append('video')
# load model
cls = diffusers.MochiPipeline
if shared.sd_model.__class__ != cls:
sd_models.unload_model_weights()
kwargs = {}
kwargs = model_quant.create_bnb_config(kwargs)
kwargs = model_quant.create_ao_config(kwargs)
shared.sd_model = cls.from_pretrained(
repo_id,
cache_dir = shared.opts.hfcache_dir,
torch_dtype=devices.dtype,
**kwargs
)
shared.sd_model.scheduler._shift = 7.0 # pylint: disable=protected-access
sd_models.set_diffuser_options(shared.sd_model)
shared.sd_model.sd_checkpoint_info = sd_checkpoint.CheckpointInfo(repo_id)
shared.sd_model.sd_model_hash = None
shared.sd_model = sd_models.apply_balanced_offload(shared.sd_model)
shared.sd_model.vae.enable_slicing()
shared.sd_model.vae.enable_tiling()
devices.torch_gc(force=True)
shared.log.debug(f'Video: cls={shared.sd_model.__class__.__name__} args={p.task_args}')
# run processing
t0 = time.time()
processed = processing.process_images(p)
t1 = time.time()
if processed is not None and len(processed.images) > 0:
shared.log.info(f'Video: frames={len(processed.images)} time={t1-t0:.2f}')
if video_type != 'None':
images.save_video(p, filename=None, images=processed.images, video_type=video_type, duration=duration, loop=gif_loop, pad=mp4_pad, interpolate=mp4_interpolate)
return processed
+1 -1
View File
@@ -81,7 +81,7 @@ class Script(scripts.Script):
p.width = 1024
p.height = 576
image = images.resize_image(resize_mode=2, im=image, width=p.width, height=p.height, upscaler_name=None, output_type='pil')
p.ops.append('svd')
p.ops.append('video')
p.do_not_save_grid = True
p.init_images = [image]
p.sampler_name = 'Default' # svd does not support non-default sampler
+1 -1
View File
@@ -87,7 +87,7 @@ class Script(scripts.Script):
shared.opts.sd_model_checkpoint = checkpoint.name
sd_models.reload_model_weights(op='model')
p.ops.append('text2video')
p.ops.append('video')
p.do_not_save_grid = True
if use_default:
p.task_args['num_frames'] = model['params'][0]
+1 -1
Submodule wiki updated: 34ba1df45d...a6c10ce38e