fix prompt enhance api with vision

Signed-off-by: Vladimir Mandic <mandic00@live.com>
This commit is contained in:
Vladimir Mandic
2026-09-10 09:03:07 +02:00
parent d23faa1f32
commit df163f3f35
9 changed files with 44 additions and 32 deletions
+17 -16
View File
@@ -1,8 +1,8 @@
# Change Log for SD.Next
## Update for 2026-09-09
## Update for 2026-09-10
### Highlights for 2026-09-09
### Highlights for 2026-09-10
*What's New*? Well, code-wise, this is a big one...
First, a-lot-of-optimizations:
@@ -23,7 +23,7 @@ Plus inevitable bug-fixes...
[Home](https://vladmandic.github.io/sdnext/) | [ChangeLog](https://github.com/vladmandic/automatic/blob/master/CHANGELOG.md) | [Docs](https://vladmandic.github.io/sdnext-docs/) | [Discord](https://discord.com/invite/sd-next-federal-batch-inspectors-1101998836328697867) | [Sponsor](https://github.com/sponsors/vladmandic)
### Details for 2026-09-09
### Details for 2026-09-10
- **Models**
- [Anima 2.9B Preview v1](https://huggingface.co/yeoj34760/Anima-2.9B)
@@ -109,24 +109,25 @@ Plus inevitable bug-fixes...
- new articles: *Attention, Modular-Pipelines*
- updated: *LoRA, MiniMax*
- **Fixes**
- prompt: unnecessary secondary prompt if same
- ui: js fetch exception handling
- api: prompt enhance with vision
- compile: keep model compiled state
- detailer: handling of stop/skip/pause
- rife: cleanup dead code, thanks @Anai-Guo
- log: ansi color handling
- lora: cleanup tags
- lucida: handle requirements
- lumina-dimoo: attention-kwargs, thanks @Anai-Guo
- network: improve type/version lookup
- lora: cleanup tags
- xyz grid: apply bool values
- vdm scheduler: fix steps, thanks @zjn20030811
- openvino: optimize recompile checks and lora loading
- log: ansi color handling
- compile: keep model compiled state
- prompt: cache checks when cfg changes
- lucida: handle requirements
- vae: fetch scale factor from the model
- todo: remove dead code, thanks @Anai-Guo
- offline: honor offline mode for more models, thanks @ryanmeador
- openvino: optimize recompile checks and lora loading
- prompt enhance: cloud models use correct system prompt
- prompt: cache checks when cfg changes
- prompt: unnecessary secondary prompt if same
- rife: cleanup dead code, thanks @Anai-Guo
- todo: remove dead code, thanks @Anai-Guo
- ui: js fetch exception handling
- vae: fetch scale factor from the model
- vdm scheduler: fix steps, thanks @zjn20030811
- xyz grid: apply bool values
## Update for 2026-08-26
+2 -1
View File
@@ -59,6 +59,7 @@ def enhance(args): # pylint: disable=redefined-outer-name
options['model'] = str(args.model)
if args.image:
options['image'] = encode(args.image)
options['use_vision'] = True
response = post('/sdapi/v1/prompt-enhance', options)
return response
@@ -72,6 +73,6 @@ if __name__ == "__main__":
parser.add_argument('--image', type=str, default=None, required=False, help='optional input image')
parser.add_argument('--nsfw', type=bool, action=argparse.BooleanOptionalAction, required=False, help='nsfw allowed')
args = parser.parse_args()
log.info(f'api-upscale: {args}')
log.info(f'api-enhance: {args}')
result = enhance(args)
log.info(result)
+1
View File
@@ -46,6 +46,7 @@ def decode_base64_to_image(encoding, quiet=False):
decoded = base64.b64decode(encoding)
data = io.BytesIO(decoded)
image = Image.open(data)
image = image.convert('RGB')
return image
except Exception as e:
log.warning(f'API cannot decode image: {e}')
+4 -4
View File
@@ -366,11 +366,11 @@ class ReqPromptEnhance(BaseModel):
repetition_penalty: Optional[float] = Field(title="Repetition penalty", default=None, description="Penalizes repeated tokens to reduce repetition (1.0=no penalty)")
top_k: Optional[int] = Field(title="Top K", default=None, description="Limits token selection to the K most likely candidates")
top_p: Optional[float] = Field(title="Top P", default=None, description="Nucleus sampling threshold (0-1)")
thinking: bool = Field(title="Thinking", default=False, description="Enable thinking/reasoning mode")
keep_thinking: bool = Field(title="Keep thinking", default=False, description="Keep thinking tokens in output")
use_vision: bool = Field(title="Use vision", default=True, description="Use vision if model supports it")
thinking: Optional[bool] = Field(title="Thinking", default=False, description="Enable thinking/reasoning mode")
keep_thinking: Optional[bool] = Field(title="Keep thinking", default=False, description="Keep thinking tokens in output")
use_vision: Optional[bool] = Field(title="Use vision", default=True, description="Use vision if model supports it")
prefill: Optional[str] = Field(title="Prefill", default=None, description="Text to prefill the model response with")
keep_prefill: bool = Field(title="Keep prefill", default=False, description="Keep prefill text in the output")
keep_prefill: Optional[bool] = Field(title="Keep prefill", default=False, description="Keep prefill text in the output")
custom_args: Optional[str] = Field(title="Custom args", default=None, description="Custom arguments for the model")
process_words: Optional[str] = Field(title="Banned words", default=None, description="List of words to process")
semantic_threshold: Optional[float] = Field(title="Semantic threshold", default=None, description="Semantic similarity threshold for processed words")
+2 -1
View File
@@ -226,6 +226,7 @@ class APIProcess:
if len(instance) == 0:
raise HTTPException(status_code=500, detail="Prompt enhancement script not found")
instance = instance[0]
decoded = decode_base64_to_image(req.image) if req.image else None
prompt = instance.enhance(
model=model,
prompt=req.prompt,
@@ -244,7 +245,7 @@ class APIProcess:
use_vision=req.use_vision,
prefill=req.prefill or '',
keep_prefill=req.keep_prefill,
image=decode_base64_to_image(req.image) if req.image else None,
image=decoded,
seed=seed,
nsfw=req.nsfw,
custom_args=req.custom_args,
+8
View File
@@ -144,6 +144,12 @@ class Options:
Keep the result faithful to the original prompt and the intended image.
Do not add unrelated concepts, non-visual commentary, or fluff.
''')
t2i_noprompt: str = textwrap.dedent('''\
You are an expert AI image prompt engineer.
Create an original, imaginative and visually rich prompt for image generation.
Be Creative!
Use concrete, image-generation language and avoid unrelated fluff or commentary.
''')
i2i_prompt: str = textwrap.dedent('''\
You are an expert AI image prompt engineer.
You will receive an image and a user prompt for editing or refinement.
@@ -159,6 +165,7 @@ class Options:
Base the prompt on what is present in the image, its style, and its atmosphere.
Use concrete, image-generation language and avoid unrelated fluff or commentary.
''')
t2v_prompt: str = textwrap.dedent('''\
You are an expert AI video prompt engineer.
You will receive a user prompt for video generation.
@@ -185,6 +192,7 @@ class Options:
Specify subject and camera movement, scene transitions, and temporal dynamics.
Use concrete, video-generation language and avoid unrelated fluff or commentary.
''')
nsfw_ok: str = textwrap.dedent('''\
NSFW, nudity and explicit content and language are allowed.
''')
+4 -3
View File
@@ -85,15 +85,16 @@ def set_template(
has_image = image is not None and isinstance(image, Image.Image)
is_video = module == 'video'
debug_log(f'Prompt enhance template: module={module} prompt={has_prompt} image={has_image} video={is_video} model="{model}" nsfw={nsfw} processor={has_processor}')
debug_log(f'Prompt enhance template: module={module} prompt={has_prompt} image={has_image} video={is_video} model="{model}" nsfw={nsfw} processor={has_processor} cloud={is_cloud_model(model)}')
"""
if has_image:
if is_cloud_model(model):
pass
elif options.processor is None:
elif not has_processor:
log.error('Prompt enhance: image not supported by model')
return prompt if prompt is not None else '' # Return original text part if image cannot be processed
"""
if has_image:
chat_template = get_image_template(system, prompt, options, nsfw, has_prompt, has_processor, is_video, image)
else:
+3 -4
View File
@@ -325,11 +325,8 @@ class PromptEnhanceScript(scripts_manager.Script):
# Strip symbols from model name if present
model = get_model_repo_from_display(model) if model else self.options.default
prompt = prompt or (self.prompt.value if self.prompt else "") # Check if self.prompt is None
image = None
if use_vision and is_vision_model(model): # handle vision toggle
image = image or self.image
if image is None:
use_vision = False
prefix = prefix or ''
suffix = suffix or ''
min_tokens = min_tokens or self.options.min_tokens
@@ -363,7 +360,9 @@ class PromptEnhanceScript(scripts_manager.Script):
# Only process images if vision is enabled and model supports it
if use_vision and is_vision_model(model):
current_image = self.get_image(image)
debug_log(f'Prompt enhance: image={current_image}')
if current_image is None:
use_vision = False
debug_log(f'Prompt enhance: image={current_image} use_vision={use_vision}')
# Check if vision was requested but no image is available
if use_vision and is_vision_model(model) and current_image is None:
+3 -3
View File
@@ -35,9 +35,6 @@ async function waitForOpts() {
const t0 = performance.now();
let t1 = performance.now();
while (true) {
if (t1 - t0 > 15000) {
log('waitForOpts delayed', t1 - t0);
}
if (t1 - t0 > 60000) {
log('waitForOpts timeout');
break;
@@ -50,6 +47,9 @@ async function waitForOpts() {
break;
}
}
if (t1 - t0 > 15000) {
log('waitForOpts delayed', Math.round(t1 - t0));
}
await sleep(100);
t1 = performance.now();
}