mirror of
https://github.com/vladmandic/automatic
synced 2026-08-25 22:20:46 +02:00
808a25749f
Signed-off-by: Vladimir Mandic <mandic00@live.com>
262 lines
14 KiB
JSON
262 lines
14 KiB
JSON
{
|
||
"Boogu Image 0.1 Turbo": {
|
||
"path": "Boogu/Boogu-Image-0.1-Turbo",
|
||
"preview": "Boogu--Boogu-Image-0.1-Turbo.jpg",
|
||
"desc": "Boogu Image 0.1 Turbo is the distilled fast inference variant of Boogu Image with the same Qwen3-VL instruction encoder and Boogu transformer architecture.",
|
||
"size": 35.81,
|
||
"date": "2026 June"
|
||
},
|
||
"Boogu Image 0.1 Edit Turbo": {
|
||
"path": "Boogu/Boogu-Image-0.1-Edit-Turbo",
|
||
"preview": "Boogu--Boogu-Image-0.1-Edit-Turbo.jpg",
|
||
"desc": "Boogu Image 0.1 Edit Turbo is the distilled editing variant of Boogu Image with motion-aware instruction encoding and fast flow-match inference.",
|
||
"size": 35.81,
|
||
"date": "2026 June"
|
||
},
|
||
"StabilityAI StableDiffusion XL Turbo": {
|
||
"path": "stabilityai/sdxl-turbo",
|
||
"preview": "stabilityai--sdxl-turbo.jpg",
|
||
"desc": "SDXL-Turbo is a fast generative text-to-image model that can synthesize photorealistic images from a text prompt in a 1-4 steps.",
|
||
"variant": "fp16",
|
||
"extras": "steps: 4, cfg_scale: 0.0",
|
||
"size": 19.38,
|
||
"date": "2023 November"
|
||
},
|
||
"Krea 2 Turbo": {
|
||
"path": "CalamitousFelicitousness/Krea-2-Turbo-Diffusers",
|
||
"preview": "CalamitousFelicitousness--Krea-2-Turbo-Diffusers.jpg",
|
||
"desc": "Krea 2 (K2) Turbo is the 8-step distilled inference model of the Krea 2 family, trained from scratch by Krea. A 12.9B-parameter single-stream flow-matching DiT that uses a Qwen3-VL-4B vision-language model as its text encoder and the Qwen-Image VAE. Runs without classifier-free guidance; LoRAs trained on Krea 2 Base apply directly.",
|
||
"extras": "sampler: Default, cfg_scale: 1.0, steps: 8",
|
||
"size": 33.5,
|
||
"date": "2026 June"
|
||
},
|
||
"StabilityAI Stable Diffusion 3.5 Turbo": {
|
||
"path": "stabilityai/stable-diffusion-3.5-large-turbo",
|
||
"variant": "fp16",
|
||
"desc": "Stable Diffusion 3.5 Large Turbo is a Multimodal Diffusion Transformer (MMDiT) text-to-image model with Adversarial Diffusion Distillation (ADD) that features improved performance in image quality, typography, complex prompt understanding, and resource-efficiency, with a focus on fewer inference steps.",
|
||
"preview": "stabilityai--stable-diffusion-3_5-large-turbo.jpg",
|
||
"extras": "sampler: Default, cfg_scale: 7.0",
|
||
"size": 36.12,
|
||
"date": "2024 October"
|
||
},
|
||
"Microsoft Lens Turbo": {
|
||
"path": "Jinstudio/Lens-Turbo",
|
||
"preview": "microsoft--Lens-Turbo.jpg",
|
||
"desc": "Microsoft Lens-Turbo is the distilled Lens variant optimized for faster text-to-image generation with fewer steps.",
|
||
"size": 28.43,
|
||
"date": "2026 May"
|
||
},
|
||
"Tencent FLUX.1 Dev SRPO": {
|
||
"path": "vladmandic/flux.1-dev-SRPO",
|
||
"preview": "vladmandic--flux.1-dev-SRPO.jpg",
|
||
"desc": "FLUX.1 Dev SRPO is Tencent trained with specific technique: Directly Aligning the Full Diffusion Trajectory with Fine-Grained Human Preference",
|
||
"extras": "sampler: Default, cfg_scale: 4.5",
|
||
"size": 31.42,
|
||
"date": "2025 September"
|
||
},
|
||
"HiDream-O1 Image Dev": {
|
||
"path": "HiDream-ai/HiDream-O1-Image-Dev",
|
||
"preview": "HiDream-ai--HiDream-O1-Image-Dev.jpg",
|
||
"desc": "HiDream-O1-Image-Dev is the distilled 8B HiDream-O1 variant tuned for 28-step fast generation using flash flow scheduling.",
|
||
"extras": "sampler: Flash, steps: 28, cfg_scale: 0.0",
|
||
"size": 35.2,
|
||
"date": "2026 May"
|
||
},
|
||
"Qwen-Image-Lightning": {
|
||
"path": "vladmandic/Qwen-Lightning",
|
||
"preview": "vladmandic--Qwen-Lightning.jpg",
|
||
"desc": "Qwen-Lightning is step-distilled from Qwen-Image to allow for generation in 8 steps.",
|
||
"extras": "steps: 8",
|
||
"size": 53.74,
|
||
"date": "2025 August"
|
||
},
|
||
"Qwen-Image-Distill": {
|
||
"path": "SahilCarterr/Qwen-Image-Distill-Full",
|
||
"preview": "SahilCarterr--Qwen-Image-Distill-Full.jpg",
|
||
"desc": "Qwen-Image-Distill is a distilled and accelerated version of Qwen-Image by DiffSynth-Studio.",
|
||
"extras": "steps: 15",
|
||
"size": 56.1,
|
||
"date": "2025 August"
|
||
},
|
||
"Baidu ERNIE-Image-Turbo": {
|
||
"path": "baidu/ERNIE-Image-Turbo",
|
||
"preview": "baidu--ERNIE-Image-Turbo.jpg",
|
||
"desc": "ERNIE-Image-Turbo is a distilled ERNIE-Image variant optimized for fast generation with fewer denoising steps.",
|
||
"extras": "sampler: Default, cfg_scale: 1.0, steps: 8",
|
||
"size": 22.29,
|
||
"date": "2026 April"
|
||
},
|
||
"Qwen-Image-Lightning-Edit": {
|
||
"path": "vladmandic/Qwen-Lightning-Edit",
|
||
"preview": "vladmandic--Qwen-Lightning-Edit.jpg",
|
||
"desc": "Qwen-Lightning-Edit is step-distilled from Qwen-Image-Edit to allow for generation in 8 steps.",
|
||
"extras": "steps: 8",
|
||
"size": 53.74,
|
||
"date": "2025 September"
|
||
},
|
||
"Qwen-Image Pruning-12B": {
|
||
"path": "OPPOer/Qwen-Image-Pruning",
|
||
"subfolder": "Qwen-Image-12B-8steps",
|
||
"preview": "OPPOer--Qwen-Image-Pruning.jpg",
|
||
"desc": "This open-source project is based on Qwen-Image and has attempted model pruning, removing 20 layers while retaining the weights of 40 layers, resulting in a model size of 12B parameters.",
|
||
"date": "2025 September",
|
||
"size": 38.54
|
||
},
|
||
"Qwen-Image-Edit Pruning-13B": {
|
||
"path": "OPPOer/Qwen-Image-Edit-Pruning",
|
||
"subfolder": "Qwen-Image-Edit-13B-4steps",
|
||
"preview": "OPPOer--Qwen-Image-Edit-Pruning.jpg",
|
||
"desc": "This open-source project is based on Qwen-Image-Edit and has attempted model pruning, removing 20 layers while retaining the weights of 40 layers, resulting in a model size of 13.6B parameters.",
|
||
"date": "2025 September",
|
||
"size": 41.08
|
||
},
|
||
"Qwen-Image-Edit-2509 Pruning-13B": {
|
||
"path": "OPPOer/Qwen-Image-Edit-2509-Pruning",
|
||
"subfolder": "Qwen-Image-Edit-2509-13B-4steps",
|
||
"preview": "OPPOer--Qwen-Image-Edit-2509-Pruning.jpg",
|
||
"desc": "This open-source project is based on Qwen-Image-Edit and has attempted model pruning, removing 20 layers while retaining the weights of 40 layers, resulting in a model size of 13.6B parameters.",
|
||
"date": "2025 October",
|
||
"size": 42.34
|
||
},
|
||
"lodestones Chroma1 Flash": {
|
||
"path": "lodestones/Chroma1-Flash",
|
||
"preview": "lodestones--Chroma1-Flash.jpg",
|
||
"desc": "Chroma is a 8.9B parameter model based on FLUX.1-schnell. It’s fully Apache 2.0 licensed, ensuring that anyone can use, modify, and build on top of it—no corporate gatekeeping. A fine-tuned version of the Chroma1-Base made to find the best way to make these flow matching models faster.",
|
||
"size": 25.6,
|
||
"date": "2025 August"
|
||
},
|
||
"SDXL Flash Mini": {
|
||
"path": "SDXL-Flash_Mini.safetensors@https://huggingface.co/sd-community/sdxl-flash-mini/resolve/main/SDXL-Flash_Mini.safetensors?download=true",
|
||
"preview": "SDXL-Flash_Mini.jpg",
|
||
"desc": "Introducing the new fast model SDXL Flash (Mini), we learned that all fast XL models work fast, but the quality decreases, and we also made a fast model, but it is not as fast as LCM, Turbo, Lightning and Hyper, but the quality is higher.",
|
||
"extras": "sampler: DEIS, steps: 40, cfg_scale: 6.0",
|
||
"experimental": true
|
||
},
|
||
"NVLabs Sana 1.5 1.6B 1k Sprint": {
|
||
"path": "Efficient-Large-Model/Sana_Sprint_1.6B_1024px_diffusers",
|
||
"desc": "SANA-Sprint is an ultra-efficient diffusion model for text-to-image (T2I) generation, reducing inference steps from 20 to 1-4 while achieving state-of-the-art performance.",
|
||
"preview": "Efficient-Large-Model--Sana15_Sprint_1600M_1024px_diffusers.jpg",
|
||
"size": 9.03,
|
||
"date": "2025 March"
|
||
},
|
||
"Segmind SSD-1B": {
|
||
"path": "huggingface/segmind/SSD-1B",
|
||
"preview": "segmind--SSD-1B.jpg",
|
||
"desc": "The Segmind Stable Diffusion Model (SSD-1B) offers a compact, efficient, and distilled version of the SDXL model. At 50% smaller and 60% faster than Stable Diffusion XL (SDXL), it provides quick and seamless performance without sacrificing image quality.",
|
||
"variant": "fp16",
|
||
"extras": "sampler: Default, cfg_scale: 9.0",
|
||
"size": 12.48,
|
||
"date": "2023 October"
|
||
},
|
||
"Segmind Tiny": {
|
||
"path": "segmind/tiny-sd",
|
||
"preview": "segmind--tiny-sd.jpg",
|
||
"desc": "Segmind's Tiny-SD offers a compact, efficient, and distilled version of Realistic Vision 4.0 and is up to 80% faster than SD1.5",
|
||
"extras": "width: 512, height: 512, sampler: Default, cfg_scale: 9.0",
|
||
"size": 0.99,
|
||
"date": "2023 July"
|
||
},
|
||
"Tencent HunyuanImage 2.1 Distilled": {
|
||
"path": "hunyuanvideo-community/HunyuanImage-2.1-Distilled-Diffusers",
|
||
"desc": "HunyuanImage-2.1, a highly efficient text-to-image model that is capable of generating 2K (2048 × 2048) resolution images.",
|
||
"preview": "hunyuanvideo-community--HunyuanImage-2.1-Distilled-Diffusers.jpg",
|
||
"size": 49.53,
|
||
"date": "2025 September"
|
||
},
|
||
"Bria Fibo-Lite": {
|
||
"path": "briaai/Fibo-lite",
|
||
"preview": "briaai--Fibo-lite.jpg",
|
||
"desc": "BRIA Fibo-lite is a lightweight, distilled variant of FIBO optimized for fast inference while maintaining strong image quality. Ideal for resource-constrained environments.",
|
||
"extras": "sampler: Default, cfg_scale: 3.5",
|
||
"size": 22.47,
|
||
"date": "2025 November"
|
||
},
|
||
"Tencent HunyuanDiT 1.2 Distilled": {
|
||
"path": "Tencent-Hunyuan/HunyuanDiT-v1.2-Diffusers-Distilled",
|
||
"desc": "Hunyuan-DiT : A Powerful Multi-Resolution Diffusion Transformer with Fine-Grained Chinese Understanding.",
|
||
"preview": "Tencent-Hunyuan--HunyuanDiT-v1.2-Diffusers-Distilled.jpg",
|
||
"extras": "sampler: Default, cfg_scale: 2.0",
|
||
"size": 13.43,
|
||
"date": "2024 July"
|
||
},
|
||
"Tencent HunyuanDiT 1.1 Distilled": {
|
||
"path": "Tencent-Hunyuan/HunyuanDiT-v1.1-Diffusers-Distilled",
|
||
"desc": "Hunyuan-DiT : A Powerful Multi-Resolution Diffusion Transformer with Fine-Grained Chinese Understanding.",
|
||
"preview": "Tencent-Hunyuan--HunyuanDiT-v1.1-Diffusers-Distilled.jpg",
|
||
"extras": "sampler: Default, cfg_scale: 2.0",
|
||
"size": 13.49,
|
||
"date": "2024 June"
|
||
},
|
||
"Black Forest Labs FLUX.2 Klein 4B": {
|
||
"path": "black-forest-labs/FLUX.2-klein-4B",
|
||
"preview": "black-forest-labs--FLUX.2-klein-4B.jpg",
|
||
"desc": "FLUX.2-klein-4B is a 4 billion parameter size-distilled version of FLUX.2-dev optimized for consumer GPUs. Achieves sub-second inference with 4 steps. Supports both text-to-image generation and multi-reference image editing. Apache 2.0 licensed.",
|
||
"extras": "sampler: Default, cfg_scale: 1.0, steps: 4",
|
||
"size": 14.87,
|
||
"date": "2026 January"
|
||
},
|
||
"Black Forest Labs FLUX.2 Klein 9B": {
|
||
"path": "black-forest-labs/FLUX.2-klein-9B",
|
||
"preview": "black-forest-labs--FLUX.2-klein-9B.jpg",
|
||
"desc": "FLUX.2-klein-9B is a 9 billion parameter size-distilled version of FLUX.2-dev. Higher quality than 4B variant with sub-second inference using 4 steps. Supports text-to-image and multi-reference editing. Non-commercial license.",
|
||
"extras": "sampler: Default, cfg_scale: 1.0, steps: 4",
|
||
"size": 32.32,
|
||
"date": "2026 January"
|
||
},
|
||
"Black Forest Labs FLUX.2 Klein 9B KV": {
|
||
"path": "black-forest-labs/FLUX.2-klein-9b-kv",
|
||
"preview": "black-forest-labs--FLUX.2-klein-9b-kv.jpg",
|
||
"desc": "FLUX.2 klein 9B KV is an optimized variant of FLUX.2 klein 9B with KV-cache support for accelerated multi-reference editing. This variant caches key-value pairs from reference images during the first denoising step, eliminating redundant computation in subsequent steps for significantly faster multi-image editing workflows.",
|
||
"extras": "sampler: Default, cfg_scale: 1.0, steps: 4",
|
||
"size": 32.32,
|
||
"date": "2026 March"
|
||
},
|
||
"Anima 1.0 Turbo": {
|
||
"path": "CalamitousFelicitousness/Anima-1.0-Turbo-Diffusers",
|
||
"preview": "CalamitousFelicitousness--Anima-1.0-Turbo-Diffusers.jpg",
|
||
"desc": "Anima 1.0 Turbo, distilled for fast generation with increased stability and a strong default style. A 2B parameter anime-focused text-to-image model based on modified Cosmos-Predict-2B with Qwen3-0.6B text encoder, created by CircleStone Labs and Comfy Org.",
|
||
"extras": "sampler: Default, cfg_scale: 1.0, steps: 10",
|
||
"date": "2026 July",
|
||
"size": 4.99
|
||
},
|
||
"Meituan LongCat Image-Edit Turbo": {
|
||
"path": "meituan-longcat/LongCat-Image-Edit-Turbo",
|
||
"preview": "meituan-longcat--LongCat-Image-Edit.jpg",
|
||
"desc": "LongCat-Image-Edit-Turbo, the distilled version of LongCat-Image-Edit. It achieves high-quality image editing with only 8 NFEs (Number of Function Evaluations) , offering extremely low inference latency.",
|
||
"size": 27.28,
|
||
"date": "2026 February"
|
||
},
|
||
"Microsoft Mage-Flow Turbo": {
|
||
"path": "vladmandic/Mage-Flow-4B-Turbo",
|
||
"preview": "vladmandic--Mage-Flow-Turbo-4B.jpg",
|
||
"desc": "Mage-Flow is a compact 4B-scale generative stack for efficient text-to-image generation and instruction-based image editing.",
|
||
"extras": "sampler: Default",
|
||
"size": 16.19,
|
||
"date": "2026 July"
|
||
},
|
||
"SeFi-Image 1B Turbo": {
|
||
"path": "SeFi-Image/SeFi-Image-1B-turbo-diffusers",
|
||
"preview": "SeFi-Image--SeFi-Image-1B-turbo-diffusers.jpg",
|
||
"desc": "SeFi-Image is a text-to-image foundation model family built with Semantic-First Diffusion. It separates generation into semantic and texture latent streams, denoising semantic structure slightly ahead of texture details.",
|
||
"extras": "sampler: Default",
|
||
"size": 6.32,
|
||
"date": "2026 July"
|
||
},
|
||
"SeFi-Image 2B Turbo": {
|
||
"path": "SeFi-Image/SeFi-Image-2B-turbo-diffusers",
|
||
"preview": "SeFi-Image--SeFi-Image-2B-turbo-diffusers.jpg",
|
||
"desc": "SeFi-Image is a text-to-image foundation model family built with Semantic-First Diffusion. It separates generation into semantic and texture latent streams, denoising semantic structure slightly ahead of texture details.",
|
||
"extras": "sampler: Default",
|
||
"size": 8.18,
|
||
"date": "2026 July"
|
||
},
|
||
"SeFi-Image 5B Turbo": {
|
||
"path": "SeFi-Image/SeFi-Image-5B-turbo-diffusers",
|
||
"preview": "SeFi-Image--SeFi-Image-5B-turbo-diffusers.jpg",
|
||
"desc": "SeFi-Image is a text-to-image foundation model family built with Semantic-First Diffusion. It separates generation into semantic and texture latent streams, denoising semantic structure slightly ahead of texture details.",
|
||
"extras": "sampler: Default",
|
||
"size": 17.69,
|
||
"date": "2026 July"
|
||
}
|
||
}
|