From 005fc5c86e9370fb1ed12a504a80eeea019fb1e9 Mon Sep 17 00:00:00 2001 From: CalamitousFelicitousness Date: Sun, 16 Aug 2026 07:34:35 +0100 Subject: [PATCH] refactor(video): accept video and audio references in the shared core The core took reference images only, so no api caller could send the video and audio references the ref2va workflow conditions on, and the marshalling that handles them existed solely in the MiniMax tab. validate_references now gates on the workflow and hands the entries to the architecture that owns them, which accepts decoded images and local file paths in any mix and preserves their order, since order fixes the labels a prompt addresses. reference_caps exposes the same limits the validation enforces, so a client reads them instead of mirroring the numbers. - MAX_IMAGE_REFERENCES is gone: the limits now cover all three kinds and a total - the run body no longer builds reference objects or knows their class - an image is converted where it is built rather than at the call site, so a reference decoded from a file and one posted as base64 arrive the same way - pipeline args summarize a reference list by kind, since a decoded video would otherwise print its frames into the per-generation log line - the video endpoint documents what it actually accepts: images alone, because video and audio decode from files rather than from the wire, and an upload reference only where an extension provides the store that resolves one --- modules/api/video.py | 13 +++++--- modules/processing_args.py | 2 ++ modules/video_models/video_run.py | 38 ++++++++++------------ test/test-video-references.py | 52 +++++++++++++++++++++++++++++++ 4 files changed, 79 insertions(+), 26 deletions(-) diff --git a/modules/api/video.py b/modules/api/video.py index 37cdf3be0..2b34357d2 100644 --- a/modules/api/video.py +++ b/modules/api/video.py @@ -27,10 +27,10 @@ class ReqVideo(BaseModel): seed: int = Field(default=-1, title="Seed", description="Generation seed; -1 for random") guidance_scale: float = Field(default=-1.0, title="Guidance scale", description="CFG scale; -1 keeps the model default") guidance_true: float = Field(default=-1.0, title="True guidance", description="True CFG scale; -1 keeps the model default") - init_image: str | None = Field(default=None, title="Init image", description="Base64, data URI, or upload reference for the first-frame image") + init_image: str | None = Field(default=None, title="Init image", description="Base64 or data URI for the first-frame image; an upload reference resolves only where an extension provides the upload store") init_strength: float = Field(default=0.8, ge=0.0, le=1.0, title="Init strength", description="Denoising strength for the init image") - last_image: str | None = Field(default=None, title="Last image", description="Base64, data URI, or upload reference for the last-frame image") - references: list[str] = Field(default=[], title="References", description="Reference images for a reference workflow, in the order the model reads them; base64, data URIs, or upload references. At most 9, each within a 1:4 to 4:1 aspect ratio. Rejected on models that do not condition on references") + last_image: str | None = Field(default=None, title="Last image", description="Base64 or data URI for the last-frame image; an upload reference resolves only where an extension provides the upload store") + references: list[str] = Field(default=[], title="References", description="Reference images for a reference workflow, in the order the model reads them; base64 or data URIs, or upload references where an extension provides the upload store. Images only: the video core also conditions on video and audio references, which this endpoint cannot carry. At most 9, each within a 1:4 to 4:1 aspect ratio. Rejected on models that do not condition on references") vae_type: str = Field(default="Default", title="VAE type", description="Decode variant: Default, Tiny, Remote, or Upscale") vae_tile_frames: int = Field(default=16, ge=1, le=64, title="VAE tile frames", description="Frames per VAE decode tile") audio: bool = Field(default=True, title="Audio", description="Generate audio on models that support it") @@ -142,11 +142,14 @@ class APIVideo: `send_thumbnail`. Artifacts above the base64 size cap return `video` empty with `video_path` set; fetch those via `GET /sdapi/v1/video/file`. - `init_image` and `last_image` accept base64 data, data URIs, or upload references. + `init_image` and `last_image` accept base64 data or data URIs. An `upload:` reference + resolves only where an extension registers an upload store; without one it is rejected. Models whose workflow is `ref2va` condition on `references` instead: an ordered list of images the prompt addresses as ``, `` and so on, following list order. A single reference may also be passed as `init_image`. Reference images do not - set the output canvas, and `last_image` is ignored. + set the output canvas, and `last_image` is ignored. The workflow also conditions on video + and audio references, addressed as `