Raw schema
{
"type": "object",
"properties": {
"prompt": {
"type": "string",
"description": "Text prompt describing the video to generate (or the edit to apply, when passing an input video)"
},
"model": {
"type": "string",
"description": "Video model id. Omit for the default (seedance-2.5 — flagship multimodal: text-to-video, image-to-video, reference images/videos/audio; pass a clip as reference_videos to edit/extend; 480p/720p/1080p, up to 30s). wan-3.0-video is the Alibaba all-in-one alternative — same modes, up to 30s, and ALL input media free (480p 60 / 720p 120 / 1080p 240 credits/sec output-only; wan-3.0-video-prime is the same model generating much faster at 85/170/340). For cheaper drafts pick seedance-2.0-mini (480p 60 / 720p 120 credits/sec). Call list_models for per-resolution prices."
},
"image": {
"type": "string",
"description": "Optional START FRAME for image-to-video — a public image URL, a data URI, or raw base64. Can't be combined with reference_images or video."
},
"end_image": {
"type": "string",
"description": "Optional END FRAME used together with image (the start frame) — the video interpolates between the two."
},
"reference_images": {
"type": "array",
"items": {
"type": "string"
},
"description": "Optional reference images guiding the video (subject consistency) — each a URL, data URI, or base64. seedance-2.5: up to 15; wan-3.0-video / wan-3.0-video-prime: up to 10 (free); seedance-2.0: up to 9 alone, 6 alongside an input video. Can't be combined with image/end_image."
},
"video": {
"type": "string",
"description": "Optional INPUT VIDEO for video-to-video editing (2-15s) — must be a URL (upload local files with upload_media first). Can't be combined with image/end_image. On the seedance-2.0 family the INPUT seconds are billed at half the output per-second rate on top of the output."
},
"audio": {
"type": "string",
"description": "Optional reference audio URL (MP3/WAV/OGG; upload local files with upload_media). Must accompany an image, reference_images, or video input. Not for minimax-h3 / seedance-2.5 / wan-3.0-video(-prime) — use reference_audios there."
},
"reference_videos": {
"type": "array",
"items": {
"type": "string"
},
"description": "minimax-h3 / seedance-2.5 / wan-3.0-video(-prime) only: reference VIDEO clip URLs guiding motion/identity — on seedance-2.5 and wan-3.0-video(-prime) also how you edit/extend a clip (h3: ≤3, 15s combined; 2.5: ≤5, 30s combined; wan-3.0(-prime): ≤5, 15s combined, and input + output ≤ 30s; upload local files with upload_media). h3 and 2.5 bill INPUT seconds on top of the output (h3 160/s; 2.5 80/160/375 per s at 480p/720p/1080p); wan-3.0-video(-prime) input is FREE. Can't be combined with image/end_image."
},
"reference_audios": {
"type": "array",
"items": {
"type": "string"
},
"description": "minimax-h3 / seedance-2.5 / wan-3.0-video(-prime) only: reference AUDIO clip URLs (WAV/MP3; h3 ≤3/15s combined, 2.5 ≤5/30s combined, wan-3.0(-prime) ≤5/15s combined; free; upload local files with upload_media). Can't be combined with image/end_image."
},
"aspect_ratio": {
"type": "string",
"description": "Aspect ratio (seedance-2.5: 16:9, 9:16, 1:1, 4:3, 3:4, 21:9 — default 16:9). On minimax-h3 it applies to TEXT-TO-VIDEO only. On seedance-2.5 it applies to t2v AND jobs with reference images/audio; a start/end frame or a reference VIDEO forces adaptive (output follows the input's frame). On wan-3.0-video (and -prime) it is honored in EVERY mode (even with a first frame or reference media) and also accepts 'adaptive' (its default — the model picks a suitable ratio from the inputs)."
},
"resolution": {
"type": "string",
"description": "Output resolution — PRICING VARIES A LOT (seedance-2.5: 480p=150, 720p=330, 1080p=750 credits per SECOND, default 720p; seedance-2.0: 1080p=550/4k=1250). Prefer 480p unless asked."
},
"duration_seconds": {
"type": "integer",
"description": "Video length in seconds (seedance-2.5: 4–30, default 5; seedance-2.0: 4–15). Cost scales linearly."
},
"negative_prompt": {
"type": "string",
"description": "What to avoid in the generated video"
},
"generate_audio": {
"type": "boolean",
"description": "Whether the model should generate audio with the video (no price impact on seedance-2.5/2.0)."
}
},
"required": [
"prompt"
],
"additionalProperties": false
}