diff --git a/CHANGELOG.md b/CHANGELOG.md index e6f00ded4..b82b0560b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -93,6 +93,7 @@ And (*as always*) many bugfixes and improvements to existing features! - refactor legacy processing loop - fix settings components mismatch - fix *Wan 2.2-5B I2V* workflow + - fix *Wan* T2I workflow - fix OpenVINO - fix video model vs pipeline mismatch - fix video generic save frames diff --git a/html/reference.json b/html/reference.json index 7988b3107..325dbf760 100644 --- a/html/reference.json +++ b/html/reference.json @@ -1,25 +1,26 @@ + { - "Tempest SD-XL v0.1": { - "path": "TempestV0.1-Artistic.safetensors@https://huggingface.co/dataautogpt3/TempestV0.1/resolve/main/TempestV0.1-Artistic.safetensors?download=true", - "preview": "TempestV0.1-Artistic.jpg", - "desc": "The TempestV0.1 Initiative is a powerhouse in image generation, leveraging an unparalleled dataset of over 6 million images. The collection's vast scale, with resolutions from 1400x2100 to 4800x7200, encompasses 200GB of high-quality content.", - "extras": "width: 2048, height: 1024, sampler: DEIS, steps: 40, cfg_scale: 6.0" + "Tempest-by-Vlad XL": { + "path": "tempestByVlad_baseV01.safetensors@https://civitai.com/api/download/models/1301775", + "preview": "tempest-by-vlad-base.jpg", + "desc": "Flexible SDXL model with custom encoder and finetuned for larger landscape resolutions with high details and high contrast.", + "extras": "" + }, + "Tempest-by-Vlad XL Hyper": { + "path": "tempestByVlad_hyperV01.safetensors@https://civitai.com/api/download/models/1343512", + "preview": "tempest-by-vlad-hyper.jpg", + "desc": "Custom distilled variant with goal to get as-normal-as-possible model that works with low steps and guidance-free", + "extras": "" }, - "Juggernaut SD-XL XI": { + "Juggernaut XL XI": { "path": "juggernautXL_juggXIByRundiffusion.safetensors@https://civitai.com/api/download/models/782002", "preview": "juggernautXL_v9Rundiffusionphoto2.jpg", "desc": "Showcase finetuned model based on Stable diffusion XL", "extras": "sampler: DEIS, steps: 20, cfg_scale: 6.0" }, - "Juggernaut SD-XL X Hyper": { - "path": "Juggernaut_X_RunDiffusion_Hyper.safetensors@https://civitai.com/api/download/models/471120", - "preview": "juggernautXL_v9Rundiffusionphoto2.jpg", - "desc": "Showcase finetuned model based on Stable diffusion XL", - "extras": "sampler: DEIS, steps: 20, cfg_scale: 6.0" - }, - "Juggernaut SD-XL IX Lightning": { - "path": "juggernautXL_v9Rdphoto2Lightning.safetensors@https://civitai.com/api/download/models/357609", + "Juggernaut XL XI Lightning": { + "path": "juggernautXL_juggXILightningByRD.safetensors@https://civitai.com/api/download/models/920957", "preview": "juggernautXL_v9Rdphoto2Lightning.jpg", "desc": "Showcase finetuned model based on Stable diffusion XL", "extras": "sampler: DPM SDE, steps: 6, cfg_scale: 2.0" @@ -32,40 +33,6 @@ "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0" }, - "DreamShaper SD v8": { - "original": true, - "path": "dreamshaper_8.safetensors@https://civitai.com/api/download/models/128713", - "preview": "dreamshaper_8.jpg", - "desc": "Showcase finetuned model based on Stable diffusion 1.5", - "extras": "width: 512, height: 512, sampler: DEIS, steps: 20, cfg_scale: 6.0" - }, - "Dreamshaper SD v7 LCM": { - "path": "SimianLuo/LCM_Dreamshaper_v7", - "preview": "SimianLuo--LCM_Dreamshaper_v7.jpg", - "desc": "Latent Consistencey Models enable swift inference with minimal steps on any pre-trained LDMs, including Stable Diffusion. By distilling classifier-free guidance into the model's input, LCM can generate high-quality images in very short inference time. LCM can generate quality images in as few as 3-4 steps, making it blazingly fast.", - "extras": "width: 512, height: 512, sampler: LCM, steps: 4, cfg_scale: 0.0" - }, - "DreamShaper SD-XL Turbo": { - "path": "dreamshaperXL_v21TurboDPMSDE.safetensors@https://civitai.com/api/download/models/351306", - "preview": "dreamshaperXL_v21TurboDPMSDE.jpg", - "desc": "Showcase finetuned model based on Stable diffusion XL", - "extras": "sampler: DPM SDE, steps: 8, cfg_scale: 2.0" - }, - - "SDXS DreamShaper 512": { - "path": "IDKiro/sdxs-512-dreamshaper", - "preview": "IDKiro--sdxs-512-dreamshaper.jpg", - "desc": "SDXS: Real-Time One-Step Latent Diffusion Models with Image Conditions", - "extras": "width: 512, height: 512, sampler: CMSI, steps: 1, cfg_scale: 0.0" - }, - "SDXL Flash Mini": { - "path": "SDXL-Flash_Mini.safetensors@https://huggingface.co/sd-community/sdxl-flash-mini/resolve/main/SDXL-Flash_Mini.safetensors?download=true", - "preview": "SDXL-Flash_Mini.jpg", - "desc": "Introducing the new fast model SDXL Flash (Mini), we learned that all fast XL models work fast, but the quality decreases, and we also made a fast model, but it is not as fast as LCM, Turbo, Lightning and Hyper, but the quality is higher.", - "extras": "width: 2048, height: 1024, sampler: DEIS, steps: 40, cfg_scale: 6.0", - "experimental": true - }, - "RunwayML StableDiffusion 1.5": { "original": true, "path": "v1-5-pruned-fp16-emaonly.safetensors@https://huggingface.co/Aptronym/SDNext/resolve/main/Reference/v1-5-pruned-fp16-emaonly.safetensors?download=true", @@ -283,6 +250,20 @@ "extras": "sampler: Default, cfg_scale: 3.5" }, + "SDXS DreamShaper 512": { + "path": "IDKiro/sdxs-512-dreamshaper", + "preview": "IDKiro--sdxs-512-dreamshaper.jpg", + "desc": "SDXS: Real-Time One-Step Latent Diffusion Models with Image Conditions", + "extras": "width: 512, height: 512, sampler: CMSI, steps: 1, cfg_scale: 0.0" + }, + "SDXL Flash Mini": { + "path": "SDXL-Flash_Mini.safetensors@https://huggingface.co/sd-community/sdxl-flash-mini/resolve/main/SDXL-Flash_Mini.safetensors?download=true", + "preview": "SDXL-Flash_Mini.jpg", + "desc": "Introducing the new fast model SDXL Flash (Mini), we learned that all fast XL models work fast, but the quality decreases, and we also made a fast model, but it is not as fast as LCM, Turbo, Lightning and Hyper, but the quality is higher.", + "extras": "width: 2048, height: 1024, sampler: DEIS, steps: 40, cfg_scale: 6.0", + "experimental": true + }, + "NVLabs Sana 1.5 1.6B 1k": { "path": "Efficient-Large-Model/SANA1.5_1.6B_1024px_diffusers", "desc": "Sana is an efficient model with scaling of training-time and inference time techniques. SANA-1.5 delivers: efficient model growth from 1.6B Sana-1.0 model to 4.8B, achieving similar or better performance than training from scratch and saving 60% training cost; efficient model depth pruning, slimming any model size as you want; powerful VLM selection based inference scaling, smaller model+inference scaling > larger model.", diff --git a/models/Reference/tempest-by-vlad-base.jpg b/models/Reference/tempest-by-vlad-base.jpg new file mode 100644 index 000000000..0d48f0a62 Binary files /dev/null and b/models/Reference/tempest-by-vlad-base.jpg differ diff --git a/models/Reference/tempest-by-vlad-hyper.jpg b/models/Reference/tempest-by-vlad-hyper.jpg new file mode 100644 index 000000000..c2ffc23e8 Binary files /dev/null and b/models/Reference/tempest-by-vlad-hyper.jpg differ diff --git a/modules/processing_args.py b/modules/processing_args.py index 5959f98f4..4cee55eb4 100644 --- a/modules/processing_args.py +++ b/modules/processing_args.py @@ -122,7 +122,7 @@ def task_specific_kwargs(p, model): 'target_subject_category': getattr(p, 'prompt', '').split()[-1], 'output_type': 'pil', } - if model.__class__.__name__ == 'StableDiffusion3Pipeline': + if model.__class__.__name__ in ['StableDiffusion3Pipeline', 'WanPipeline']: p.width = 16 * (p.width // 16) p.height = 16 * (p.height // 16) task_args['width'] = p.width diff --git a/modules/processing_diffusers.py b/modules/processing_diffusers.py index d809362e3..8343559c4 100644 --- a/modules/processing_diffusers.py +++ b/modules/processing_diffusers.py @@ -458,7 +458,8 @@ def validate_pipeline(p: processing.StableDiffusionProcessing): if m.repo_cls is not None: models_cls.append(m.repo_cls.__name__) is_video_model = shared.sd_model.__class__.__name__ in models_cls - is_video_pipeline = 'video' in p.__class__.__name__.lower() + override_video_pipelines = ['WanPipeline'] + is_video_pipeline = ('video' in p.__class__.__name__.lower()) or (shared.sd_model.__class__.__name__ in override_video_pipelines) if is_video_model and not is_video_pipeline: shared.log.error(f'Mismatch: type={shared.sd_model_type} cls={shared.sd_model.__class__.__name__} request={p.__class__.__name__} video model with non-video pipeline') return False diff --git a/modules/prompt_parser_xhinker.py b/modules/prompt_parser_xhinker.py index fed488c2a..377a9f3cc 100644 --- a/modules/prompt_parser_xhinker.py +++ b/modules/prompt_parser_xhinker.py @@ -439,13 +439,13 @@ def get_weighted_text_embeddings_sdxl( , pad_last_block=pad_last_block ) - prompt_token_groups_2, prompt_weight_groups_2 = group_tokens_and_weights( + prompt_token_groups_2, _prompt_weight_groups_2 = group_tokens_and_weights( prompt_tokens_2.copy() , prompt_weights_2.copy() , pad_last_block=pad_last_block ) - neg_prompt_token_groups_2, neg_prompt_weight_groups_2 = group_tokens_and_weights( + neg_prompt_token_groups_2, _neg_prompt_weight_groups_2 = group_tokens_and_weights( neg_prompt_tokens_2.copy() , neg_prompt_weights_2.copy() , pad_last_block=pad_last_block @@ -609,7 +609,6 @@ def get_weighted_text_embeddings_sdxl_refiner( , generator = torch.Generator(text2img_pipe.device).manual_seed(2) ).images[0] """ - import math eos = 49407 # pipe.tokenizer.eos_token_id # tokenizer 2 @@ -1148,13 +1147,13 @@ def get_weighted_text_embeddings_sd3( , pad_last_block=pad_last_block ) - prompt_token_groups_2, prompt_weight_groups_2 = group_tokens_and_weights( + prompt_token_groups_2, _prompt_weight_groups_2 = group_tokens_and_weights( prompt_tokens_2.copy() , prompt_weights_2.copy() , pad_last_block=pad_last_block ) - neg_prompt_token_groups_2, neg_prompt_weight_groups_2 = group_tokens_and_weights( + neg_prompt_token_groups_2, _neg_prompt_weight_groups_2 = group_tokens_and_weights( neg_prompt_tokens_2.copy() , neg_prompt_weights_2.copy() , pad_last_block=pad_last_block @@ -1374,7 +1373,7 @@ def get_weighted_text_embeddings_flux1( pipe.tokenizer_2, prompt2 ) - prompt_token_groups, prompt_weight_groups = group_tokens_and_weights( + prompt_token_groups, _prompt_weight_groups = group_tokens_and_weights( prompt_tokens.copy() , prompt_weights.copy() , pad_last_block=True