mirror of
https://github.com/vladmandic/automatic
synced 2026-09-19 09:14:35 +02:00
Merge pull request #4553 from CalamitousFelicitousness/feat/flux2-klein-support
Feat/flux2 klein support
This commit is contained in:
@@ -161,5 +161,25 @@
|
||||
"preview": "Tencent-Hunyuan--HunyuanDiT-v1.1-Diffusers-Distilled.jpg",
|
||||
"tags": "distilled",
|
||||
"extras": "sampler: Default, cfg_scale: 2.0"
|
||||
},
|
||||
"Black Forest Labs FLUX.2 Klein 4B": {
|
||||
"path": "black-forest-labs/FLUX.2-klein-4B",
|
||||
"preview": "black-forest-labs--FLUX.2-klein-4B.jpg",
|
||||
"desc": "FLUX.2-klein-4B is a 4 billion parameter size-distilled version of FLUX.2-dev optimized for consumer GPUs. Achieves sub-second inference with 4 steps. Supports both text-to-image generation and multi-reference image editing. Apache 2.0 licensed.",
|
||||
"skip": true,
|
||||
"tags": "distilled",
|
||||
"extras": "sampler: Default, cfg_scale: 4.0, steps: 4",
|
||||
"size": 8.5,
|
||||
"date": "2025 January"
|
||||
},
|
||||
"Black Forest Labs FLUX.2 Klein 9B": {
|
||||
"path": "black-forest-labs/FLUX.2-klein-9B",
|
||||
"preview": "black-forest-labs--FLUX.2-klein-9B.jpg",
|
||||
"desc": "FLUX.2-klein-9B is a 9 billion parameter size-distilled version of FLUX.2-dev. Higher quality than 4B variant with sub-second inference using 4 steps. Supports text-to-image and multi-reference editing. Non-commercial license.",
|
||||
"skip": true,
|
||||
"tags": "distilled",
|
||||
"extras": "sampler: Default, cfg_scale: 4.0, steps: 4",
|
||||
"size": 18.5,
|
||||
"date": "2025 January"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+19
-1
@@ -124,11 +124,29 @@
|
||||
"size": 104.74,
|
||||
"date": "2025 November"
|
||||
},
|
||||
"Black Forest Labs FLUX.2 Klein Base 4B": {
|
||||
"path": "black-forest-labs/FLUX.2-klein-base-4B",
|
||||
"preview": "black-forest-labs--FLUX.2-klein-base-4B.jpg",
|
||||
"desc": "FLUX.2-klein-base-4B is the undistilled 4 billion parameter base model of FLUX.2-klein. Requires 50 inference steps for full quality but offers flexibility for fine-tuning. Supports text-to-image and multi-reference editing. Apache 2.0 licensed.",
|
||||
"skip": true,
|
||||
"extras": "sampler: Default, cfg_scale: 4.0, steps: 50",
|
||||
"size": 8.5,
|
||||
"date": "2025 January"
|
||||
},
|
||||
"Black Forest Labs FLUX.2 Klein Base 9B": {
|
||||
"path": "black-forest-labs/FLUX.2-klein-base-9B",
|
||||
"preview": "black-forest-labs--FLUX.2-klein-base-9B.jpg",
|
||||
"desc": "FLUX.2-klein-base-9B is the undistilled 9 billion parameter base model of FLUX.2-klein. Requires 50 inference steps for full quality but offers flexibility for fine-tuning. Supports text-to-image and multi-reference editing. Non-commercial license.",
|
||||
"skip": true,
|
||||
"extras": "sampler: Default, cfg_scale: 4.0, steps: 50",
|
||||
"size": 18.5,
|
||||
"date": "2025 January"
|
||||
},
|
||||
|
||||
"Z-Image-Turbo": {
|
||||
"path": "Tongyi-MAI/Z-Image-Turbo",
|
||||
"preview": "Tongyi-MAI--Z-Image-Turbo.jpg",
|
||||
"desc": "Z-Image-Turbo, a distilled version of Z-Image that matches or exceeds leading competitors with only 8 NFEs (Number of Function Evaluations). It offers sub-second inference latency on enterprise-grade H800 GPUs and fits comfortably within 16G VRAM consumer devices. It excels in photorealistic image generation, bilingual text rendering (English & Chinese), and robust instruction adherence.",
|
||||
"desc": "Z-Image-Turbo, a distilled version of Z-Image that matches or exceeds leading competitors with only 8 NFEs (Number of Function Evaluations). It excels in photorealistic image generation, bilingual text rendering (English & Chinese), and robust instruction adherence.",
|
||||
"skip": true,
|
||||
"extras": "sampler: Default, cfg_scale: 1.0, steps: 9",
|
||||
"size": 20.3,
|
||||
|
||||
+1
-1
@@ -648,7 +648,7 @@ def check_diffusers():
|
||||
t_start = time.time()
|
||||
if args.skip_all:
|
||||
return
|
||||
sha = '5efb81fa711863fdece9136ad10788440e658b40' # diffusers commit hash
|
||||
sha = '61f175660a8ac54f1470a74a810e6c38fb4795d5' # diffusers commit hash
|
||||
# if args.use_rocm or args.use_zluda or args.use_directml:
|
||||
# sha = '043ab2520f6a19fce78e6e060a68dbc947edb9f9' # lock diffusers versions for now
|
||||
pkg = pkg_resources.working_set.by_key.get('diffusers', None)
|
||||
|
||||
@@ -116,7 +116,7 @@ def diffusers_callback(pipe, step: int = 0, timestep: int = 0, kwargs: dict = {}
|
||||
if current_noise_pred is None:
|
||||
current_noise_pred = kwargs.get("predicted_image_embedding", None)
|
||||
|
||||
if hasattr(pipe, "_unpack_latents") and hasattr(pipe, "vae_scale_factor"): # FLUX
|
||||
if hasattr(pipe, "_unpack_latents") and hasattr(pipe, "vae_scale_factor"): # FLUX.1
|
||||
if p.hr_resize_mode > 0 and (p.hr_upscaler != 'None' or p.hr_resize_mode == 5) and p.is_hr_pass:
|
||||
width = max(getattr(p, 'width', 0), getattr(p, 'hr_upscale_to_x', 0))
|
||||
height = max(getattr(p, 'height', 0), getattr(p, 'hr_upscale_to_y', 0))
|
||||
@@ -128,6 +128,37 @@ def diffusers_callback(pipe, step: int = 0, timestep: int = 0, kwargs: dict = {}
|
||||
shared.state.current_noise_pred = pipe._unpack_latents(current_noise_pred, height, width, pipe.vae_scale_factor) # pylint: disable=protected-access
|
||||
else:
|
||||
shared.state.current_noise_pred = current_noise_pred
|
||||
elif hasattr(pipe, "_unpatchify_latents"): # FLUX.2 - unpack [B, seq, patch_ch] to [B, ch, H, W]
|
||||
# Get dimensions for unpacking, same logic as FLUX.1
|
||||
vae_scale = getattr(pipe, 'vae_scale_factor', 8)
|
||||
if p.hr_resize_mode > 0 and (p.hr_upscaler != 'None' or p.hr_resize_mode == 5) and p.is_hr_pass:
|
||||
width = max(getattr(p, 'width', 0), getattr(p, 'hr_upscale_to_x', 0))
|
||||
height = max(getattr(p, 'height', 0), getattr(p, 'hr_upscale_to_y', 0))
|
||||
else:
|
||||
width = getattr(p, 'width', 1024)
|
||||
height = getattr(p, 'height', 1024)
|
||||
latents = kwargs['latents']
|
||||
if len(latents.shape) == 3: # packed format [B, seq_len, patch_channels]
|
||||
b, seq_len, patch_ch = latents.shape
|
||||
channels = patch_ch // 4 # 4 = 2x2 patch
|
||||
h_patches = height // vae_scale // 2
|
||||
w_patches = width // vae_scale // 2
|
||||
if h_patches * w_patches != seq_len: # fallback to square assumption
|
||||
h_patches = w_patches = int(seq_len ** 0.5)
|
||||
# [B, h*w, C*4] -> [B, h, w, C, 2, 2] -> [B, C, h, 2, w, 2] -> [B, C, H, W]
|
||||
latents = latents.view(b, h_patches, w_patches, channels, 2, 2)
|
||||
latents = latents.permute(0, 3, 1, 4, 2, 5).reshape(b, channels, h_patches * 2, w_patches * 2)
|
||||
shared.state.current_latent = latents
|
||||
if current_noise_pred is not None and len(current_noise_pred.shape) == 3:
|
||||
b, seq_len, patch_ch = current_noise_pred.shape
|
||||
channels = patch_ch // 4
|
||||
h_patches = height // vae_scale // 2
|
||||
w_patches = width // vae_scale // 2
|
||||
if h_patches * w_patches != seq_len:
|
||||
h_patches = w_patches = int(seq_len ** 0.5)
|
||||
current_noise_pred = current_noise_pred.view(b, h_patches, w_patches, channels, 2, 2)
|
||||
current_noise_pred = current_noise_pred.permute(0, 3, 1, 4, 2, 5).reshape(b, channels, h_patches * 2, w_patches * 2)
|
||||
shared.state.current_noise_pred = current_noise_pred
|
||||
else:
|
||||
shared.state.current_latent = kwargs['latents']
|
||||
shared.state.current_noise_pred = current_noise_pred
|
||||
|
||||
@@ -92,6 +92,8 @@ def guess_by_name(fn, current_guess):
|
||||
new_guess = 'HiDream'
|
||||
elif 'chroma' in fn.lower() and 'xl' not in fn.lower():
|
||||
new_guess = 'Chroma'
|
||||
elif 'flux.2' in fn.lower() and 'klein' in fn.lower():
|
||||
new_guess = 'FLUX2 Klein'
|
||||
elif 'flux.2' in fn.lower():
|
||||
new_guess = 'FLUX2'
|
||||
elif 'flux' in fn.lower() or 'flex.1' in fn.lower():
|
||||
|
||||
@@ -359,6 +359,10 @@ def load_diffuser_force(detected_model_type, checkpoint_info, diffusers_load_con
|
||||
from pipelines.model_flux2 import load_flux2
|
||||
sd_model = load_flux2(checkpoint_info, diffusers_load_config)
|
||||
allow_post_quant = False
|
||||
elif model_type in ['FLUX2 Klein']:
|
||||
from pipelines.model_flux2_klein import load_flux2_klein
|
||||
sd_model = load_flux2_klein(checkpoint_info, diffusers_load_config)
|
||||
allow_post_quant = False
|
||||
elif model_type in ['FLEX']:
|
||||
from pipelines.model_flex import load_flex
|
||||
sd_model = load_flex(checkpoint_info, diffusers_load_config)
|
||||
|
||||
+21
-3
@@ -12,11 +12,17 @@ import torch
|
||||
from modules import devices, paths, shared
|
||||
|
||||
|
||||
debug = os.environ.get('SD_PREVIEW_DEBUG', None) is not None
|
||||
|
||||
|
||||
TAESD_MODELS = {
|
||||
'TAESD 1.3 Mocha Croissant': { 'fn': 'taesd_13_', 'uri': 'https://github.com/madebyollin/taesd/raw/7f572ca629c9b0d3c9f71140e5f501e09f9ea280', 'model': None },
|
||||
'TAESD 1.2 Chocolate-Dipped Shortbread': { 'fn': 'taesd_12_', 'uri': 'https://github.com/madebyollin/taesd/raw/8909b44e3befaa0efa79c5791e4fe1c4d4f7884e', 'model': None },
|
||||
'TAESD 1.1 Fruit Loops': { 'fn': 'taesd_11_', 'uri': 'https://github.com/madebyollin/taesd/raw/3e8a8a2ab4ad4079db60c1c7dc1379b4cc0c6b31', 'model': None },
|
||||
'TAESD 1.0': { 'fn': 'taesd_10_', 'uri': 'https://github.com/madebyollin/taesd/raw/88012e67cf0454e6d90f98911fe9d4aef62add86', 'model': None },
|
||||
'TAE FLUX.1': { 'fn': 'taef1.pth', 'uri': 'https://github.com/madebyollin/taesd/raw/main/taef1_decoder.pth', 'model': None },
|
||||
'TAE FLUX.2': { 'fn': 'taef2.pth', 'uri': 'https://github.com/madebyollin/taesd/raw/main/taef2_decoder.pth', 'model': None },
|
||||
'TAE SD3': { 'fn': 'taesd3.pth', 'uri': 'https://github.com/madebyollin/taesd/raw/main/taesd3_decoder.pth', 'model': None },
|
||||
'TAE HunyuanVideo': { 'fn': 'taehv.pth', 'uri': 'https://github.com/madebyollin/taehv/raw/refs/heads/main/taehv.pth', 'model': None },
|
||||
'TAE WanVideo': { 'fn': 'taew1.pth', 'uri': 'https://github.com/madebyollin/taehv/raw/refs/heads/main/taew2_1.pth', 'model': None },
|
||||
'TAE MochiVideo': { 'fn': 'taem1.pth', 'uri': 'https://github.com/madebyollin/taem1/raw/refs/heads/main/taem1.pth', 'model': None },
|
||||
@@ -38,7 +44,7 @@ prev_cls = ''
|
||||
prev_type = ''
|
||||
prev_model = ''
|
||||
lock = threading.Lock()
|
||||
supported = ['sd', 'sdxl', 'sd3', 'f1', 'h1', 'zimage', 'lumina2', 'hunyuanvideo', 'wanai', 'chrono', 'cosmos', 'mochivideo', 'pixartsigma', 'pixartalpha', 'hunyuandit', 'omnigen', 'qwen', 'longcat', 'omnigen2', 'flite', 'ovis', 'kandinsky5', 'glmimage', 'cogview3', 'cogview4']
|
||||
supported = ['sd', 'sdxl', 'sd3', 'f1', 'f2', 'h1', 'zimage', 'lumina2', 'hunyuanvideo', 'wanai', 'chrono', 'cosmos', 'mochivideo', 'pixartsigma', 'pixartalpha', 'hunyuandit', 'omnigen', 'qwen', 'longcat', 'omnigen2', 'flite', 'ovis', 'kandinsky5', 'glmimage', 'cogview3', 'cogview4']
|
||||
|
||||
|
||||
def warn_once(msg, variant=None):
|
||||
@@ -59,8 +65,14 @@ def get_model(model_type = 'decoder', variant = None):
|
||||
model_cls = 'sd'
|
||||
elif model_cls in {'pixartsigma', 'hunyuandit', 'omnigen', 'auraflow'}:
|
||||
model_cls = 'sdxl'
|
||||
elif model_cls in {'h1', 'zimage', 'lumina2', 'chroma', 'longcat', 'omnigen2', 'flite', 'ovis', 'kandinsky5', 'glmimage', 'cogview3', 'cogview4'}:
|
||||
elif model_cls in {'f1', 'h1', 'zimage', 'lumina2', 'chroma', 'longcat', 'omnigen2', 'flite', 'ovis', 'kandinsky5', 'glmimage', 'cogview3', 'cogview4'}:
|
||||
model_cls = 'f1'
|
||||
variant = 'TAE FLUX.1'
|
||||
elif model_cls == 'f2':
|
||||
model_cls = 'f2'
|
||||
variant = 'TAE FLUX.2'
|
||||
elif model_cls == 'sd3':
|
||||
variant = 'TAE SD3'
|
||||
elif model_cls in {'wanai', 'qwen', 'chrono', 'cosmos'}:
|
||||
variant = variant or 'TAE WanVideo'
|
||||
elif model_cls not in supported:
|
||||
@@ -149,7 +161,13 @@ def decode(latents):
|
||||
dtype = devices.dtype_vae if devices.dtype_vae != torch.bfloat16 else torch.float16 # taesd does not support bf16
|
||||
tensor = latents.unsqueeze(0) if len(latents.shape) == 3 else latents
|
||||
tensor = tensor.detach().clone().to(devices.device, dtype=dtype)
|
||||
if variant.startswith('TAESD'):
|
||||
if debug:
|
||||
shared.log.debug(f'Decode: type="taesd" variant="{variant}" input={latents.shape} tensor={tensor.shape}')
|
||||
# Fallback: reshape packed 128-channel latents to 32 channels if not already unpacked
|
||||
if variant == 'TAE FLUX.2' and len(tensor.shape) == 4 and tensor.shape[1] == 128:
|
||||
b, _c, h, w = tensor.shape
|
||||
tensor = tensor.reshape(b, 32, h * 2, w * 2)
|
||||
if variant.startswith('TAESD') or variant in {'TAE FLUX.1', 'TAE FLUX.2', 'TAE SD3'}:
|
||||
image = vae.decoder(tensor).clamp(0, 1).detach()
|
||||
image = image[0]
|
||||
else:
|
||||
|
||||
@@ -48,6 +48,8 @@ pipelines = {
|
||||
'Qwen': getattr(diffusers, 'QwenImagePipeline', None),
|
||||
'HunyuanImage': getattr(diffusers, 'HunyuanImagePipeline', None),
|
||||
'Z-Image': getattr(diffusers, 'ZImagePipeline', None),
|
||||
'FLUX2': getattr(diffusers, 'Flux2Pipeline', None),
|
||||
'FLUX2 Klein': getattr(diffusers, 'Flux2KleinPipeline', None),
|
||||
'LongCat': getattr(diffusers, 'LongCatImagePipeline', None),
|
||||
'GLM-Image': getattr(diffusers, 'GlmImagePipeline', None),
|
||||
# dynamically imported and redefined later
|
||||
|
||||
@@ -77,7 +77,11 @@ class TAESD(nn.Module): # pylint: disable=abstract-method
|
||||
self.decoder = self.decoder.to(devices.device, dtype=self.dtype)
|
||||
|
||||
def guess_latent_channels(self, decoder_path, encoder_path):
|
||||
return 16 if ("f1" in encoder_path or "f1" in decoder_path) or ("sd3" in encoder_path or "sd3" in decoder_path) else 4
|
||||
if "f2" in encoder_path or "f2" in decoder_path:
|
||||
return 32 # FLUX.2 uses 32 latent channels
|
||||
if ("f1" in encoder_path or "f1" in decoder_path) or ("sd3" in encoder_path or "sd3" in decoder_path):
|
||||
return 16
|
||||
return 4
|
||||
|
||||
@staticmethod
|
||||
def scale_latents(x):
|
||||
|
||||
@@ -200,6 +200,23 @@ def load_text_encoder(repo_id, cls_name, load_config=None, subfolder="text_encod
|
||||
**load_args,
|
||||
**quant_args,
|
||||
)
|
||||
# Qwen3ForCausalLM - shared text encoders by hidden_size:
|
||||
# - Z-Image, Klein-4B: Qwen3-4B (hidden_size=2560)
|
||||
# - Klein-9B: Qwen3-8B (hidden_size=4096)
|
||||
elif cls_name == transformers.Qwen3ForCausalLM and allow_shared and shared.opts.te_shared_t5:
|
||||
if '-9b' in repo_id.lower():
|
||||
shared_repo = 'black-forest-labs/FLUX.2-klein-9B' # 9B variants use Qwen3-8B
|
||||
else:
|
||||
shared_repo = 'Tongyi-MAI/Z-Image-Turbo' # 4B variants and Z-Image use Qwen3-4B
|
||||
subfolder = 'text_encoder'
|
||||
shared.log.debug(f'Load model: text_encoder="{shared_repo}" cls={cls_name.__name__} quant="{quant_type}" loader={_loader("transformers")} shared={shared.opts.te_shared_t5}')
|
||||
text_encoder = cls_name.from_pretrained(
|
||||
shared_repo,
|
||||
cache_dir=shared.opts.hfcache_dir,
|
||||
subfolder=subfolder,
|
||||
**load_args,
|
||||
**quant_args,
|
||||
)
|
||||
|
||||
# load from repo
|
||||
if text_encoder is None:
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
import transformers
|
||||
import diffusers
|
||||
from modules import shared, devices, sd_models, model_quant, sd_hijack_te, sd_hijack_vae
|
||||
from pipelines import generic
|
||||
|
||||
|
||||
def load_flux2_klein(checkpoint_info, diffusers_load_config=None):
|
||||
if diffusers_load_config is None:
|
||||
diffusers_load_config = {}
|
||||
repo_id = sd_models.path_to_repo(checkpoint_info)
|
||||
sd_models.hf_auth_check(checkpoint_info)
|
||||
|
||||
load_args, _quant_args = model_quant.get_dit_args(diffusers_load_config, allow_quant=False)
|
||||
shared.log.debug(f'Load model: type=Flux2Klein repo="{repo_id}" config={diffusers_load_config} offload={shared.opts.diffusers_offload_mode} dtype={devices.dtype} args={load_args}')
|
||||
|
||||
# Load transformer - Klein uses Flux2Transformer2DModel (same class as Flux2, different size)
|
||||
transformer = generic.load_transformer(repo_id, cls_name=diffusers.Flux2Transformer2DModel, load_config=diffusers_load_config)
|
||||
|
||||
# Load text encoder - Klein uses Qwen3 (4B for Klein-4B, 8B for Klein-9B)
|
||||
text_encoder = generic.load_text_encoder(repo_id, cls_name=transformers.Qwen3ForCausalLM, load_config=diffusers_load_config)
|
||||
|
||||
pipe = diffusers.Flux2KleinPipeline.from_pretrained(
|
||||
repo_id,
|
||||
transformer=transformer,
|
||||
text_encoder=text_encoder,
|
||||
cache_dir=shared.opts.diffusers_dir,
|
||||
**load_args,
|
||||
)
|
||||
pipe.task_args = {
|
||||
'output_type': 'np',
|
||||
}
|
||||
diffusers.pipelines.auto_pipeline.AUTO_TEXT2IMAGE_PIPELINES_MAPPING["flux2klein"] = diffusers.Flux2KleinPipeline
|
||||
diffusers.pipelines.auto_pipeline.AUTO_IMAGE2IMAGE_PIPELINES_MAPPING["flux2klein"] = diffusers.Flux2KleinPipeline
|
||||
diffusers.pipelines.auto_pipeline.AUTO_INPAINT_PIPELINES_MAPPING["flux2klein"] = diffusers.Flux2KleinPipeline
|
||||
|
||||
del text_encoder
|
||||
del transformer
|
||||
sd_hijack_te.init_hijack(pipe)
|
||||
sd_hijack_vae.init_hijack(pipe)
|
||||
|
||||
devices.torch_gc(force=True, reason='load')
|
||||
return pipe
|
||||
Reference in New Issue
Block a user