Files
2026-03-03 12:03:54 +05:30

18708 lines
623 KiB
JSON

[
{
"name": "ai-video-effects",
"category": "Image to Video",
"variant": "AI Video Effects",
"family": "effects",
"group_of": "ai-effects",
"description": "AI Video Effects applies advanced visual transformations, color grading, and cinematic filters to create stunning videos from images.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to insert into the predefined prompt template for the selected effect.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"name": {
"enum": [
"360 Rotation",
"Abandoned Places",
"Angry",
"Animal Documentary",
"Assassin It",
"Baby It",
"Boxing",
"Bride It",
"Cakeify",
"Cartoon Jaw Drop",
"Cats",
"Crush It",
"Crying",
"Cyberpunk 2077",
"Deflate It",
"Disney Princess It",
"Dogs",
"Eye Close-Up",
"Fantasy Landscapes",
"Film Noir",
"Fire",
"Glamor",
"Goblin",
"Gun Reveal",
"Hug Jesus",
"Hulk Transformation",
"Inflate It",
"Jungle It",
"Jumpscare",
"Kamehameha",
"Kiss Cam",
"Kissing",
"Lego",
"Laughing",
"Little Planet",
"Live Wallpaper",
"Looping Pixel Art",
"Melt It",
"Mona Lisa It",
"Museum It",
"Muscle Show Off",
"Orc",
"Pixar",
"Pirate Captain",
"POV Driving",
"Princess It",
"Puppy it",
"Robotic Face Reveal",
"Samurai It",
"Sharingan Eyes",
"Skyrim Fus-Ro-Dah",
"Snow White It",
"Squish It",
"Steamboat Willie",
"Super Saiyan Transformation",
"Tsunami",
"Ultra Wide",
"VHS Footage",
"VIP It",
"Warrior It",
"Wind Blast",
"Younger Self Selfie",
"Zen It",
"Zoom Call"
],
"description": "The type of effect to apply to the video.",
"type": "string",
"title": "Effect Type",
"name": "name",
"default": "Cakeify"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "generate_wan_ai_effects"
}
}
}
},
{
"name": "motion-controls",
"category": "Image to Video",
"variant": "Motion Controls",
"family": "effects",
"group_of": "ai-effects",
"description": "Motion Controls adds dynamic camera movements, speed ramps, and zoom effects to bring your images to life as smooth, engaging videos.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to insert into the predefined prompt template for the selected effect.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"name": {
"enum": [
"360 Orbit",
"Arc Shot",
"Car Chase",
"Car Mount Cam",
"Crash Zoom In",
"Crash Zoom Out",
"Crane Down",
"Crane Overhead",
"Crane Punch-In",
"Crane Up",
"Dirty Lens",
"Dolly In",
"Dolly Left",
"Dolly Out",
"Dolly Right",
"Dolly Zoom In",
"Dolly Zoom Out",
"Dutch Angle",
"Fast Dolly Zoom In",
"Fast Dolly Zoom Out",
"Fisheye Lens",
"Focus Shift",
"FPV Drone Cam",
"Handheld Cam",
"Head Tracking",
"Hero Run",
"Human Timelapse",
"Landscape Timelapse",
"Lazy Susan",
"Lens Crac",
"Lens Flare",
"Matrix Shot",
"Motion Blur",
"Object POV",
"Overhead",
"Rap Video Cam",
"Robotic Cam",
"Snorricam",
"Tilt Down",
"Tilt Up",
"Whip Pan",
"Wiggle",
"Zoom In",
"Zoom In Through Object",
"Zoom Into Mouth",
"Zoom Out",
"Zoom Out Through Object"
],
"description": "The type of effect to apply to the video.",
"type": "string",
"title": "Effect Type",
"name": "name",
"default": "360 Orbit"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "generate_wan_ai_effects"
}
}
}
},
{
"name": "vfx",
"category": "Image to Video",
"variant": "VFX",
"family": "effects",
"group_of": "ai-effects",
"description": "VFX delivers high-impact visual effects like explosions, particles, and cinematic overlays to transform static images into action-packed videos.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to insert into the predefined prompt template for the selected effect.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"name": {
"enum": [
"Building Explosion",
"Car Explosion",
"Decay Time-Lapse",
"Disintegration",
"Electricity",
"Flying",
"Huge Explosion",
"Levitate",
"Tornado"
],
"description": "The type of effect to apply to the video.",
"type": "string",
"title": "Effect Type",
"name": "name",
"default": "Car Explosion"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "generate_wan_ai_effects"
}
}
}
},
{
"name": "veo3-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "veo",
"group_of": "video",
"description": "VEO3 I2V animates static images into expressive video sequences, adding lifelike movement while preserving the original composition.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the desired video content.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide start frame image. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 1
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "veo3-image-to-video"
}
}
}
},
{
"name": "veo3-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "veo",
"group_of": "video",
"description": "VEO3 T2V generates cinematic videos from text prompts, capturing dynamic motion, rich scenes, and storytelling visuals in stunning detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the desired video content.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "veo3-text-to-video"
}
}
}
},
{
"name": "veo3-fast-text-to-video",
"category": "Text to Video",
"variant": "Text to Video [Fast]",
"family": "veo",
"group_of": "video",
"description": "VEO3 Fast T2V creates short videos from text instantly, balancing speed and quality for quick content generation and prototyping.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the desired video content.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "veo3-fast-text-to-video"
}
}
}
},
{
"name": "ai-image-upscaler",
"category": "Image to Image",
"variant": "Image Upscaler",
"family": "tools",
"group_of": "ai-tools",
"description": "Transform blurry or pixelated images into high-definition visuals. Our AI Image Upscaler uses deep learning to reconstruct details and bring your visuals to life.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "ai-image-upscale"
}
}
}
},
{
"name": "ai-image-face-swap",
"category": "Image to Image",
"variant": "Image Faceswap",
"family": "tools",
"group_of": "ai-tools",
"description": "Advanced facial recognition and blending algorithms enable precise face swaps while preserving skin tone, lighting, and facial geometry.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "Target URL of the input image. Image width and height cannot be greater than 2048px",
"field": "image",
"type": "string",
"title": "Target Image",
"name": "target_url"
},
"swap_url": {
"description": "Swap URL of the input image. Image width and height cannot be greater than 2048px",
"field": "image",
"type": "string",
"title": "Swap Image",
"name": "swap_url"
},
"target_index": {
"title": "Target Index",
"name": "target_index",
"type": "int",
"description": "0 = largest face. To switch to another target face - switch to index 1.",
"default": 0,
"minValue": 0,
"maxValue": 10,
"step": 1
}
},
"title": "BaseInput",
"required": [
"image_url",
"swap_url"
],
"endpoint_url": "ai-image-face-swap"
}
}
}
},
{
"name": "ai-video-face-swap",
"category": "Video to Video",
"variant": "Video Faceswap",
"family": "tools",
"group_of": "ai-tools",
"description": "Replace faces in videos with stunning realism. Our AI ensures accurate expression transfer, lighting consistency, and smooth frame-by-frame blending.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "Swap URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"video_url": {
"description": "Target URL of the input video. Resolution must be less than 1280x720 or 720x1280. Size must be less than 10mb. Video frame cannot be greater than 600",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"target_gender": {
"enum": [
"All",
"Female",
"Male"
],
"description": "Gender filter for target faces to be swapped. 'All' means no filter.",
"type": "string",
"title": "Target Gender",
"name": "target_gender",
"default": "All"
},
"target_index": {
"title": "Target Index",
"name": "target_index",
"type": "int",
"description": "0 = largest face. To switch to another target face - switch to index 1.",
"default": 0,
"minValue": 0,
"maxValue": 10,
"step": 1
}
},
"title": "BaseInput",
"required": [
"image_url",
"video_url"
],
"endpoint_url": "ai-video-face-swap"
}
}
}
},
{
"name": "ai-dress-change",
"category": "Image to Image",
"variant": "Dress Change",
"family": "tools",
"group_of": "ai-tools",
"description": "Instantly change outfits in images using AI. Visualize different clothing styles without the need for physical trials\u2014perfect for fashion, e-commerce, and virtual try-ons.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"model_image_url": {
"description": "Model URL of the input image.",
"field": "image",
"type": "string",
"title": "Model Image URL",
"name": "model_image_url"
},
"garment_image_url": {
"description": "Garments URL of the input image.",
"field": "image",
"type": "string",
"title": "Garment Image URL",
"name": "garment_image_url"
}
},
"title": "BaseInput",
"required": [
"model_image_url",
"garment_image_url"
],
"endpoint_url": "ai-dress-change"
}
}
}
},
{
"name": "mmaudio-v2-text-to-audio",
"category": "Text to Audio",
"variant": "v2 Text to Audio",
"family": "mmaudio",
"group_of": "ai-tools",
"description": "Convert text into natural-sounding speech using mmAudio-v2. Ideal for voiceovers, virtual assistants, and content narration with lifelike clarity and tone.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the audio for.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the audio to generate.",
"default": 8,
"minValue": 1,
"maxValue": 30,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "mmaudio-v2/text-to-audio"
}
}
}
},
{
"name": "mmaudio-v2-video-to-video",
"category": "Video to Video",
"variant": "v2 Video to Video",
"family": "mmaudio",
"group_of": "ai-tools",
"description": "MMAudio-v2 generates high-quality, synchronized audio from video or text inputs. Seamlessly integrate it with AI video models to create fully-voiced, expressive video content.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the audio for.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "The URL of the video to generate the audio for.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the audio to generate.",
"default": 8,
"minValue": 1,
"maxValue": 30,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "mmaudio-v2/video-to-video"
}
}
}
},
{
"name": "ai-background-remover",
"category": "Image to Image",
"variant": "Background Remover",
"family": "tools",
"group_of": "ai-tools",
"description": "Instantly remove image backgrounds with pixel-perfect precision. Ideal for product photos, profile pictures, and creative projects.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "ai-background-remover"
}
}
}
},
{
"name": "ai-product-shot",
"category": "Image to Image",
"variant": "Product Shot",
"family": "tools",
"group_of": "ai-tools",
"description": "Instantly generate studio-quality product images with AI. Upload your item photo and get clean, stylized shots perfect for e-commerce, ads, and catalogs.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"scene_description": {
"description": "Text description of the new scene or background for the provided product shot. Bria currently supports prompts in English only, excluding special characters.",
"type": "string",
"title": "Scene Description",
"name": "scene_description"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"scene_description",
"image_url"
],
"endpoint_url": "ai-product-shot"
}
}
}
},
{
"name": "ai-skin-enhancer",
"category": "Image to Image",
"variant": "Skin Enhancer",
"family": "tools",
"group_of": "ai-tools",
"description": "Smooth skin, reduce blemishes, and enhance complexion with natural-looking results. Perfect for portraits, selfies, and professional photo retouching.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "ai-skin-enhancer"
}
}
}
},
{
"name": "ai-color-photo",
"category": "Image to Image",
"variant": "Color Photo",
"family": "tools",
"group_of": "ai-tools",
"description": "Automatically add lifelike colors to black-and-white images. Our AI brings history to life with natural tones, accurate shading, and context-aware colorization.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "ai-color-photo"
}
}
}
},
{
"name": "flux-dev",
"category": "Text to Image",
"variant": "Dev",
"family": "flux",
"group_of": "image",
"description": "Generate stunning visuals from simple text prompts. Flux Dev transforms your ideas into high-quality, creative images using powerful AI vision models. Perfect for design, storytelling, concept art, and marketing.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-dev-image"
}
}
}
},
{
"name": "veo3-fast-image-to-video",
"category": "Image to Video",
"variant": "Image to Video [Fast]",
"family": "veo",
"group_of": "video",
"description": "Quickly transform static images into short, motion-rich video clips with fast rendering and impressive quality \u2014 powered by Google's VEO3 on MuAPI.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the desired video content.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide start frame image. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 1
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "veo3-fast-image-to-video"
}
}
}
},
{
"name": "flux-dev-lora",
"category": "Training",
"variant": "Dev LoRA",
"family": "flux",
"group_of": "image",
"description": "Enables text-to-image generation using custom LoRA models. Generate consistent characters, styles, or branded visuals with high quality and fast results.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"model_id": {
"title": "LoRA Ids",
"name": "model_id",
"type": "array",
"items": {
"type": "object",
"properties": {
"model": {
"type": "string",
"format": "url",
"title": "Model ID",
"description": "The Civitai LoRA model ID."
},
"weight": {
"type": "number",
"title": "Weight",
"description": "A list of LoRA models to use for generation. Each item must include an `id` (e.g., \"civitai:1642876@1864626\") and a `weight` between 0 and 4. You can include up to 4 models. The `id` can be found in the Civitai model URL. These models will be applied with the specified weights by the Flux Dev system during image generation.",
"minValue": 0,
"maxValue": 4,
"step": 0.01,
"default": 1
}
}
},
"description": "The unique identifier of a LoRA model hosted on Civitai, used by the Flux Dev image generation system. This ID tells Flux Dev which specific LoRA model to apply during generation. You can find the model ID in the Civitai model URL (e.g., model_id: civitai:1642876@1864626).",
"maxItems": 4
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64,
"isEdit": true
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64,
"isEdit": true
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"model_id"
],
"endpoint_url": "flux_dev_lora_image"
}
}
}
},
{
"name": "flux-kontext-dev-i2i",
"category": "Image to Image",
"variant": "Kontext Dev I2I",
"family": "kontext",
"group_of": "image",
"description": "Takes an input images and transforms it based on a new prompt. Keeps structure or pose while changing style, appearance, or details.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 10
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"3:2",
"2:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-kontext-dev-i2i"
}
}
}
},
{
"name": "flux-kontext-dev-t2i",
"category": "Text to Image",
"variant": "Kontext Dev T2I",
"family": "kontext",
"group_of": "image",
"description": "Generates an image from a text prompt, with optional reference image for pose or style guidance. Ideal for controlled, consistent image creation using just a description.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"3:2",
"2:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-kontext-dev-t2i"
}
}
}
},
{
"name": "hidream-i1-fast",
"category": "Text to Image",
"variant": "Fast",
"family": "hidream",
"group_of": "image",
"description": "Optimized for speed, this variant generates images in just a few steps. Ideal for previews, real-time applications, and use cases where fast results are more important than fine detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "hidream_i1_fast_image"
}
}
}
},
{
"name": "hidream-i1-dev",
"category": "Text to Image",
"variant": "Dev",
"family": "hidream",
"group_of": "image",
"description": "Optimized for speed, this variant generates images in just a few steps. Ideal for previews, real-time applications, and use cases where fast results are more important than fine detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "hidream_i1_dev_image"
}
}
}
},
{
"name": "hidream-i1-full",
"category": "Text to Image",
"variant": "Full",
"family": "hidream",
"group_of": "image",
"description": "The most advanced version of HiDream I1, delivering high-resolution, detailed images with superior prompt understanding. Best suited for production, content creation, and high-fidelity applications.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "hidream_i1_full_image"
}
}
}
},
{
"name": "ai-product-photography",
"category": "Image to Image",
"variant": "Product Photography",
"family": "tools",
"group_of": "ai-tools",
"description": "Create professional-grade product photos using AI. Upload your item image and describe it with a prompt, and get studio-style, lifestyle, or creative backgrounds in seconds",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"person_image_url": {
"description": "Person URL of the input image.",
"field": "image",
"type": "string",
"title": "Person Image URL",
"name": "person_image_url"
},
"product_image_url": {
"description": "Product URL of the input image.",
"field": "image",
"type": "string",
"title": "Product Image URL",
"name": "product_image_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"person_image_url",
"product_image_url"
],
"endpoint_url": "ai-product-photography"
}
}
}
},
{
"name": "ai-ghibli-style",
"category": "Image to Image",
"variant": "Ghibli Style",
"family": "tools",
"group_of": "ai-tools",
"description": "Bring your imagination to life with art inspired by the enchanting world of Studio Ghibli. This AI model generates dreamy, hand-drawn visuals with soft colors, whimsical characters, and painterly backgrounds",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "ai-ghibli-style"
}
}
}
},
{
"name": "ai-anime-generator",
"category": "Text to Image",
"variant": "Anime Generator",
"family": "tools",
"group_of": "ai-tools",
"description": "Create stunning anime-style artwork instantly with our AI Anime Generator. Customize characters, scenes, and styles effortlessly in seconds!",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1,
"isEdit": true
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "ai-anime-generator"
}
}
}
},
{
"name": "ai-image-extension",
"category": "Image to Image",
"variant": "Image Extension",
"family": "tools",
"group_of": "ai-tools",
"description": "Expand the edges of any image with AI. This model continues your original photo or artwork beyond its borders while matching style, lighting, and content.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "ai-image-extension"
}
}
}
},
{
"name": "ai-object-eraser",
"category": "Image to Image",
"variant": "Object Eraser",
"family": "tools",
"group_of": "ai-tools",
"description": "Easily remove unwanted objects, people, or text from any image using AI. Just select the area you want to erase, and the model will intelligently fill the space with realistic background matching the surrounding environment. No Photoshop skills needed.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image to erase from.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"mask_image_url": {
"description": "The URL of the binary mask image that represents the area that will be cleaned.",
"field": "image",
"type": "string",
"title": "Mask URL",
"name": "mask_image_url"
}
},
"title": "BaseInput",
"required": [
"image_url",
"mask_image_url"
],
"endpoint_url": "ai-object-eraser"
}
}
}
},
{
"name": "runway-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "runway",
"group_of": "video",
"description": "Animate any image by turning it into a video with motion effects or scene continuity. RunwayML\u2019s I2V model transforms static visuals into short clips by extrapolating depth, movement, and temporal dynamics.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to be used to generate a video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image. Provides visual reference for AI",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video. If 1080p is selected, 8-second video cannot be generated.",
"default": "720p"
},
"duration": {
"enum": [
5,
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration in seconds. If 8-second video is selected, 1080p resolution cannot be used.",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "runway-image-to-video"
}
}
}
},
{
"name": "runway-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "runway",
"group_of": "video",
"description": "Generate short, high-quality videos from plain text prompts. RunwayML\u2019s text-to-video model interprets your written description and animates it into a moving visual scene with realistic or stylized motion.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to be used to generate a video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video. If 1080p is selected, 8-second video cannot be generated.",
"default": "720p"
},
"duration": {
"enum": [
5,
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration in seconds. If 8-second video is selected, 1080p resolution cannot be used.",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "runway-text-to-video"
}
}
}
},
{
"name": "suno-create-music",
"category": "Text to Audio",
"variant": "Create Music",
"family": "suno",
"group_of": "music",
"description": "Suno generate music that turns text prompts into full songs \u2014 complete with vocals, lyrics, and instrumentation. You can describe a mood, genre, or even a specific lyric idea, and Suno creates a realistic, studio-quality track in seconds.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "A description of the desired audio content. The prompt will be strictly used as the lyrics and sung in the generated track",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"style": {
"description": "Music style specification for the generated audio.",
"format": "text",
"type": "string",
"title": "Style",
"name": "style",
"placeholder": "Jazz, Classical, Electronic, Pop, Rock, Hip-hop, etc."
},
"model": {
"enum": [
"V3_5",
"V4",
"V4_5",
"V4_5PLUS",
"V5"
],
"title": "Model",
"name": "model",
"type": "string",
"description": "The AI model version to use for generation.",
"default": "V5"
},
"instrumental": {
"type": "boolean",
"title": "Instrumental",
"name": "instrumental",
"description": "Enable this option to generate music without prompt. If false prompt will used as the exact lyrics.",
"default": true
},
"negative_tags": {
"title": "Negative Tags",
"name": "negative_tags",
"type": "string",
"format": "text",
"description": "Music styles or traits to exclude from the generated audio (optional). Use to avoid specific styles.",
"placeholder": "Heavy Metal, Upbeat Drums"
},
"vocal_gender": {
"enum": [
"male",
"female"
],
"title": "Vocal Gender",
"name": "vocal_gender",
"type": "string",
"description": "Vocal gender preference for the singing voice (optional).",
"default": null
},
"style_weight": {
"title": "Style Weight",
"name": "style_weight",
"type": "int",
"description": "Strength of adherence to the specified style (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"weirdness_constraint": {
"title": "Weirdness Constraint",
"name": "weirdness_constraint",
"type": "int",
"description": "Controls experimental/creative deviation (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"audio_weight": {
"title": "Audio Weight",
"name": "audio_weight",
"type": "int",
"description": "Balance weight for audio features vs. other factors (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
}
},
"title": "BaseInput",
"required": [
"style"
],
"endpoint_url": "suno-create-music"
}
}
}
},
{
"name": "suno-remix-music",
"category": "Text to Audio",
"variant": "Remix Music",
"family": "suno",
"group_of": "music",
"description": "This API covers an audio track by transforming it into a new style while retaining its core melody. It incorporates Suno's upload capability, enabling users to upload an audio file for processing. The expected result is a refreshed audio track with a new style, keeping the original melody intact.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "A description of the desired audio content. The prompt will be strictly used as the lyrics and sung in the generated track. Maximum 3000 characters",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"audio_url": {
"description": "The URL for uploading audio files. Ensure the uploaded audio does not exceed 2 minutes in length.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"style": {
"description": "Music style specification for the generated audio.",
"format": "text",
"type": "string",
"title": "Style",
"name": "style",
"placeholder": "Jazz, Classical, Electronic, Pop, Rock, Hip-hop, etc."
},
"model": {
"enum": [
"V3_5",
"V4",
"V4_5",
"V4_5PLUS",
"V5"
],
"title": "Model",
"name": "model",
"type": "string",
"description": "The AI model version to use for generation.",
"default": "V5"
},
"instrumental": {
"type": "boolean",
"title": "Instrumental",
"name": "instrumental",
"description": "Enable this option to generate music without prompt. If false prompt will used as the exact lyrics.",
"default": true
},
"negative_tags": {
"title": "Negative Tags",
"name": "negative_tags",
"type": "string",
"format": "text",
"description": "Music styles or traits to exclude from the generated audio (optional). Use to avoid specific styles.",
"placeholder": "Heavy Metal, Upbeat Drums"
},
"vocal_gender": {
"enum": [
"male",
"female"
],
"title": "Vocal Gender",
"name": "vocal_gender",
"type": "string",
"description": "Vocal gender preference for the singing voice (optional).",
"default": null
},
"style_weight": {
"title": "Style Weight",
"name": "style_weight",
"type": "int",
"description": "Strength of adherence to the specified style (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"weirdness_constraint": {
"title": "Weirdness Constraint",
"name": "weirdness_constraint",
"type": "int",
"description": "Controls experimental/creative deviation (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"audio_weight": {
"title": "Audio Weight",
"name": "audio_weight",
"type": "int",
"description": "Balance weight for audio features vs. other factors (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
}
},
"title": "BaseInput",
"required": [
"audio_url",
"style"
],
"endpoint_url": "suno-remix-music"
}
}
}
},
{
"name": "suno-extend-music",
"category": "Text to Audio",
"variant": "Extend Music",
"family": "suno",
"group_of": "music",
"description": "This API extends audio tracks while preserving the original style of the audio track. It includes Suno's upload functionality, allowing users to upload audio files for processing. The expected result is a longer track that seamlessly continues the input style.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "A description of the desired audio content. The prompt will be strictly used as the lyrics and sung in the generated track. Maximum 3000 characters",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"audio_url": {
"description": "The URL for uploading audio files. Ensure the uploaded audio does not exceed 2 minutes in length.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"style": {
"description": "Music style specification for the generated audio.",
"format": "text",
"type": "string",
"title": "Style",
"name": "style",
"placeholder": "Jazz, Classical, Electronic, Pop, Rock, Hip-hop, etc."
},
"model": {
"enum": [
"V3_5",
"V4",
"V4_5",
"V4_5PLUS",
"V5"
],
"title": "Model",
"name": "model",
"type": "string",
"description": "The AI model version to use for generation.",
"default": "V5"
},
"continue_at": {
"title": "Continue At",
"name": "continue_at",
"type": "int",
"description": "The time point (in seconds) from which to start extending the music. Value range: greater than 0 and less than the total duration of the uploaded audio. Specifies the position in the original track where the extension should begin.",
"default": 1,
"minValue": 1,
"maxValue": 60,
"step": 1
},
"instrumental": {
"type": "boolean",
"title": "Instrumental",
"name": "instrumental",
"description": "Enable this option to generate music without prompt. If false prompt will used as the exact lyrics.",
"default": true
},
"negative_tags": {
"title": "Negative Tags",
"name": "negative_tags",
"type": "string",
"format": "text",
"description": "Music styles or traits to exclude from the generated audio (optional). Use to avoid specific styles.",
"placeholder": "Heavy Metal, Upbeat Drums"
},
"vocal_gender": {
"enum": [
"male",
"female"
],
"title": "Vocal Gender",
"name": "vocal_gender",
"type": "string",
"description": "Vocal gender preference for the singing voice (optional).",
"default": null
},
"style_weight": {
"title": "Style Weight",
"name": "style_weight",
"type": "int",
"description": "Strength of adherence to the specified style (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"weirdness_constraint": {
"title": "Weirdness Constraint",
"name": "weirdness_constraint",
"type": "int",
"description": "Controls experimental/creative deviation (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"audio_weight": {
"title": "Audio Weight",
"name": "audio_weight",
"type": "int",
"description": "Balance weight for audio features vs. other factors (optional). Range 0\u20131, up to 2 decimal places.",
"minValue": 0,
"maxValue": 1,
"step": 0.01
}
},
"title": "BaseInput",
"required": [
"prompt",
"audio_url",
"style"
],
"endpoint_url": "suno-extend-music"
}
}
}
},
{
"name": "wan2.1-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "wan2.1",
"group_of": "image",
"description": "WAN 2.1 is a powerful AI model that transforms text prompts into high-resolution, photorealistic images. It excels at detailed object rendering, realistic lighting, and fine textures, making it ideal for visual content, concept art, advertising, and digital storytelling.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.1-text-to-image"
}
}
}
},
{
"name": "flux-kontext-pro-t2i",
"category": "Text to Image",
"variant": "Kontext Pro T2I",
"family": "kontext",
"group_of": "image",
"description": "Flux Kontext Pro T2I offers fast and reliable generation with creative flexibility. It supports stylized prompts, character design, and fantasy themes while maintaining clear subject coherence.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"16:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-kontext-pro-t2i"
}
}
}
},
{
"name": "flux-kontext-pro-i2i",
"category": "Image to Image",
"variant": "Kontext Pro I2I",
"family": "kontext",
"group_of": "image",
"description": "Flux Kontext Pro I2I variant enables transforming base images into refined artwork while keeping structure intact. It\u2019s useful for sketch refinement, visual style changes, and creative edits such as re-dressing, relighting, or re-theming with prompt guidance.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 2
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"16:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-kontext-pro-i2i"
}
}
}
},
{
"name": "flux-kontext-max-t2i",
"category": "Text to Image",
"variant": "Kontext Max T2I",
"family": "kontext",
"group_of": "image",
"description": "Flux Kontext Max T2I delivers photorealistic or cinematic-quality images with exceptional detail. It's optimized for high-end visuals \u2014 from realistic humans to polished product renders.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"16:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-kontext-max-t2i"
}
}
}
},
{
"name": "flux-kontext-max-i2i",
"category": "Image to Image",
"variant": "Kontext Max I2I",
"family": "kontext",
"group_of": "image",
"description": "Flux Kontext Max I2I in Max mode allows precise image enhancement and visual transformations while retaining the source layout. It\u2019s powerful for retouching, photo-to-art workflows, concept refinement.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 2
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"16:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-kontext-max-i2i"
}
}
}
},
{
"name": "gpt4o-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "gpt",
"group_of": "image",
"description": "Generate images from text prompts using GPT-4o's vision capabilities. Ideal for basic concept visuals, diagrams, and abstract compositions.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"enum": [
1,
2,
4
],
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "gpt4o-text-to-image"
}
}
}
},
{
"name": "gpt4o-image-to-image",
"category": "Image to Image",
"variant": "Image to Image",
"family": "gpt",
"group_of": "image",
"description": "Transform an input image based on a new prompt \u2014 like changing style, lighting, or composition. Useful for reinterpreting visuals while keeping structure.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 5
},
"aspect_ratio": {
"enum": [
"1:1",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"enum": [
1,
2,
4
],
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "gpt4o-image-to-image"
}
}
}
},
{
"name": "gpt4o-edit",
"category": "Image to Image",
"variant": "Edit Image",
"family": "gpt",
"group_of": "image",
"description": "Edit a specific part of an image using natural language. Ideal for object removal, replacement, or content-aware filling.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image to erase from.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"mask_image_url": {
"description": "The URL of the binary mask image that represents the area that will be cleaned.",
"field": "image",
"type": "string",
"title": "Mask URL",
"name": "mask_image_url"
},
"aspect_ratio": {
"enum": [
"1:1",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"enum": [
1,
2,
4
],
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url",
"mask_image_url"
],
"endpoint_url": "gpt4o-edit"
}
}
}
},
{
"name": "wan2.1-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "wan2.1",
"group_of": "video",
"description": "Animate static images into expressive video sequences with WAN 2.1. Upload any image and guide its transformation into a moving scene \u2014 great for bringing art, characters, or photos to life with smooth motion and consistent style.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "wan2.1-image-to-video"
}
}
}
},
{
"name": "wan2.1-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "wan2.1",
"group_of": "video",
"description": "WAN 2.1 turns your written prompts into vivid, cinematic video clips. Ideal for storytelling, content creation, and visualizing abstract ideas, it supports detailed natural scenes, character motion, and dramatic camera movements \u2014 all from just text.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.1-text-to-video"
}
}
}
},
{
"name": "midjourney-v7-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "midjourney",
"group_of": "image",
"description": "Midjourney V7 produces high-quality, stylized images from text prompts. Known for its artistic flair, surreal composition, and vivid textures, it's perfect for character concepts, fantasy environments, and creative illustrations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the image",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"speed": {
"enum": [
"relaxed",
"fast",
"turbo"
],
"title": "Speed",
"name": "speed",
"type": "string",
"description": "The speed of which corresponds to different speed of Midjourney",
"default": "relaxed"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"1:2",
"2:1",
"2:3",
"3:2",
"5:6",
"6:5"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"variety": {
"title": "Variety",
"name": "variety",
"type": "int",
"description": "Controls the diversity of generated images. Increment by 5 each time. Higher values create more diverse results. Lower values create more consistent results.",
"default": 5,
"minValue": 0,
"maxValue": 100,
"step": 5
},
"stylization": {
"title": "Stylization",
"name": "stylization",
"type": "int",
"description": "Controls the artistic style intensity. Higher values create more stylized results. Lower values create more realistic results.",
"default": 1,
"minValue": 0,
"maxValue": 1000,
"step": 1
},
"weirdness": {
"title": "Weirdness",
"name": "weirdness",
"type": "int",
"description": "Controls the creativity and uniqueness. Higher values create more unusual results. Lower values create more conventional results.",
"default": 1,
"minValue": 0,
"maxValue": 3000,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "midjourney-v7-text-to-image"
}
}
}
},
{
"name": "midjourney-v7-image-to-image",
"category": "Image to Image",
"variant": "Image to Image",
"family": "midjourney",
"group_of": "image",
"description": "Use Midjourney V7\u2019s I2I to refine or reinterpret existing images. Modify style, mood, lighting, or content while preserving the overall composition \u2014 great for alternate versions, art variations, or polishing concepts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the image",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"speed": {
"enum": [
"relaxed",
"fast",
"turbo"
],
"title": "Speed",
"name": "speed",
"type": "string",
"description": "The speed of which corresponds to different speed of Midjourney",
"default": "relaxed"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"1:2",
"2:1",
"2:3",
"3:2",
"5:6",
"6:5"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"variety": {
"title": "Variety",
"name": "variety",
"type": "int",
"description": "Controls the diversity of generated images. Increment by 5 each time. Higher values create more diverse results. Lower values create more consistent results.",
"default": 5,
"minValue": 0,
"maxValue": 100,
"step": 5
},
"stylization": {
"title": "Stylization",
"name": "stylization",
"type": "int",
"description": "Controls the artistic style intensity. Higher values create more stylized results. Lower values create more realistic results.",
"default": 1,
"minValue": 0,
"maxValue": 1000,
"step": 1
},
"weirdness": {
"title": "Weirdness",
"name": "weirdness",
"type": "int",
"description": "Controls the creativity and uniqueness. Higher values create more unusual results. Lower values create more conventional results.",
"default": 1,
"minValue": 0,
"maxValue": 3000,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "midjourney-v7-image-to-image"
}
}
}
},
{
"name": "midjourney-v7-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "midjourney",
"group_of": "video",
"description": "Midjourney V7\u2019s I2V breathes motion into still images, animating characters, environments, and objects with artistic transitions. Ideal for looping visual stories, concept animations, or enhancing still visuals with subtle motion.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"1:2",
"2:1",
"2:3",
"3:2",
"5:6",
"6:5"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"480p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"num_videos": {
"enum": [
1,
2,
4
],
"title": "Number of videos",
"name": "num_videos",
"type": "int",
"description": "Number of videos generated in single request. Each number will charge separately",
"default": 1
},
"variety": {
"title": "Variety",
"name": "variety",
"type": "int",
"description": "Controls the diversity of generated images. Increment by 5 each time. Higher values create more diverse results. Lower values create more consistent results.",
"default": 5,
"minValue": 0,
"maxValue": 100,
"step": 5
},
"stylization": {
"title": "Stylization",
"name": "stylization",
"type": "int",
"description": "Controls the artistic style intensity. Higher values create more stylized results. Lower values create more realistic results.",
"default": 1,
"minValue": 0,
"maxValue": 1000,
"step": 1
},
"weirdness": {
"title": "Weirdness",
"name": "weirdness",
"type": "int",
"description": "Controls the creativity and uniqueness. Higher values create more unusual results. Lower values create more conventional results.",
"default": 1,
"minValue": 0,
"maxValue": 3000,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "midjourney-v7-image-to-video"
}
}
}
},
{
"name": "wan2.1-lora-i2v",
"category": "Training",
"variant": "Image to Video (LoRA)",
"family": "wan2.1",
"group_of": "video",
"description": "Bring still images to life using WAN 2.1 LoRA I2V, which supports custom LoRA fine-tunes for identity consistency. Animate expressions, subtle movements, or full-body actions while preserving personalized features from the image and LoRA.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt for generating the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"lora_list": {
"title": "LoRA List",
"name": "lora_list",
"type": "array",
"items": {
"type": "object",
"properties": {
"path": {
"type": "string",
"title": "Path",
"format": "url",
"name": "path",
"description": "URL or the path to the LoRA weights."
},
"scale": {
"type": "number",
"title": "Scale",
"name": "scale",
"description": "The scale of the LoRA weight. This is used to scale the LoRA weight before merging it with the base model. Default value: 1",
"minValue": 0,
"maxValue": 4,
"step": 0.01,
"default": 1
}
}
},
"description": "The LoRA weights for generating video",
"maxItems": 3
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "wan2.1-lora-i2v"
}
}
}
},
{
"name": "wan2.1-lora-t2v",
"category": "Training",
"variant": "Text to Video (LoRA)",
"family": "wan2.1",
"group_of": "video",
"description": "WAN 2.1 LoRA T2V enables users to generate videos from text prompts with custom-trained LoRA modules. Tailor the generation to specific characters, outfits, or animation styles \u2014 ideal for brand storytelling, fan content, and stylized animations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt for generating the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"lora_list": {
"title": "LoRA List",
"name": "lora_list",
"type": "array",
"items": {
"type": "object",
"properties": {
"path": {
"type": "string",
"title": "Path",
"format": "url",
"name": "path",
"description": "URL or the path to the LoRA weights."
},
"scale": {
"type": "number",
"title": "Scale",
"name": "scale",
"description": "The scale of the LoRA weight. This is used to scale the LoRA weight before merging it with the base model. Default value: 1",
"minValue": 0,
"maxValue": 4,
"step": 0.01,
"default": 1
}
}
},
"description": "The LoRA weights for generating video",
"maxItems": 3
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.1-lora-t2v"
}
}
}
},
{
"name": "hunyuan-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "hunyuan",
"group_of": "video",
"description": "Hunyuan I2V takes a static image and generates realistic video animations by interpreting motion and context. It works well for human portraits, objects, or scenes, adding lifelike movement while maintaining the image's integrity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "hunyuan-image-to-video"
}
}
}
},
{
"name": "hunyuan-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "hunyuan",
"group_of": "video",
"description": "Hunyuan T2V generates detailed and dynamic videos from text prompts with a focus on realism and coherent motion. It handles multi-object scenes, human actions, and cinematic compositions effectively, making it ideal for storytelling and visual concepts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "hunyuan-text-to-video"
}
}
}
},
{
"name": "hunyuan-fast-text-to-video",
"category": "Text to Video",
"variant": "Fast Text to Video",
"family": "hunyuan",
"group_of": "video",
"description": "Hunyuan Fast T2V provides accelerated video generation from text prompts with slightly reduced detail but excellent speed. Ideal for rapid prototyping, concept testing, and short-form ideas where time is critical.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "hunyuan-fast-text-to-video"
}
}
}
},
{
"name": "flux-schnell",
"category": "Text to Image",
"variant": "Schnell",
"family": "flux",
"group_of": "image",
"description": "Flux Schnell is a lightning-fast image generation model designed for rapid iterations. It delivers good visual quality from text prompts almost instantly, making it perfect for real-time concept testing, brainstorming, and UI-integrated experiences.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image. The value must be divisible by 64, eg: 128...512, 576, 640...2048.",
"default": 1024,
"minValue": 128,
"maxValue": 2048,
"step": 64
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-schnell-image"
}
}
}
},
{
"name": "seedance-lite-i2v",
"category": "Image to Video",
"variant": "Lite Image to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Lite I2V version animates static images into short videos quickly, focusing on basic motion effects and efficient processing\u2014best suited for fast demos or mobile-friendly use.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "seedance-lite-i2v"
}
}
}
},
{
"name": "seedance-lite-t2v",
"category": "Text to Video",
"variant": "Lite Text to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Lite T2V offers quick video generation from text with decent visual quality and motion. Ideal for fast previews, prototyping, or lightweight use cases where speed matters more than fine detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-lite-t2v"
}
}
}
},
{
"name": "seedance-pro-i2v",
"category": "Image to Video",
"variant": "Pro Image to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Pro I2V advanced model animates still images into stunning short videos, preserving intricate visual details and applying smooth motion dynamics, ideal for high-end visuals and cinematic edits.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "seedance-pro-i2v"
}
}
}
},
{
"name": "seedance-pro-t2v",
"category": "Text to Video",
"variant": "Pro Text to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Pro delivers high-fidelity video generation from text, producing rich visuals, smooth camera movement, and realistic scenes. Best for storytelling, content creation, and visual production.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-pro-t2v"
}
}
}
},
{
"name": "bytedance-seedream-v3",
"category": "Text to Image",
"variant": "Text to Image v3",
"family": "seedream",
"group_of": "image",
"description": "Seedream is designed for generating visually rich and artistic images from text prompts. It excels at fantasy, anime, surrealism, and vibrant color compositions \u2014 ideal for creative visuals, storyboards, and concept art.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "bytedance-seedream-image"
}
}
}
},
{
"name": "bytedance-seededit-v3",
"category": "Image to Image",
"variant": "Edit Image v3",
"family": "seedream",
"group_of": "image",
"description": "Seededit allows precise edits to images using masks and prompt guidance. Whether you're replacing backgrounds, changing clothing, or inpainting missing areas, Seededit ensures realistic, high-quality results with semantic control.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image to used to transform the image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "bytedance-seededit-image"
}
}
}
},
{
"name": "kling-v2.1-master-t2v",
"category": "Text to Video",
"variant": "Master Text to Video",
"family": "kling-v2.1",
"group_of": "video",
"description": "Kling 2.1 Master\u2019s T2V mode allows users to generate vivid, high-quality videos from detailed text prompts. It supports dynamic scenes, natural motion, and cinematic quality \u2014 perfect for storytelling, ads, or content creation from imagination alone.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-v2.1-master-t2v"
}
}
}
},
{
"name": "kling-v2.1-master-i2v",
"category": "Image to Video",
"variant": "Master Image to Video",
"family": "kling-v2.1",
"group_of": "video",
"description": "Kling 2.1 Master\u2019s I2V animates a still image into a coherent video sequence. It interprets motion, environment, and context to create realistic, visually stunning video outputs \u2014 ideal for animating portraits, scenes, or concept art.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v2.1-master-i2v"
}
}
}
},
{
"name": "kling-v2.1-standard-i2v",
"category": "Image to Video",
"variant": "Standard Image to Video",
"family": "kling-v2.1",
"group_of": "video",
"description": "Kling 2.1 Standard (developed by Kuaishou) brings static images to life by generating smooth, realistic video clips from a single frame. It captures subtle motion, background dynamics, and camera movement to produce professional-looking animations \u2014 ideal for portraits, digital art, and cinematic illustrations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v2.1-standard-i2v"
}
}
}
},
{
"name": "kling-v2.1-pro-i2v",
"category": "Image to Video",
"variant": "Pro Image to Video",
"family": "kling-v2.1",
"group_of": "video",
"description": "Kling 2.1 Pro is the high-end version of Kuaishou\u2019s video generation model, offering enhanced realism, longer motion sequences, and cinematic quality. In I2V mode, it animates static images with fluid environmental effects.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v2.1-pro-i2v"
}
}
}
},
{
"name": "wan2.2-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "wan2.2",
"group_of": "video",
"description": "Wan 2.2\u2019s I2V mode brings static visuals to life with vivid, expressive animations. It interprets motion, emotion, and background dynamics from a single image to generate smooth and cinematic short videos.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 5,
"minValue": 5,
"maxValue": 8,
"step": 3
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "wan2.2-image-to-video"
}
}
}
},
{
"name": "wan2.2-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "wan2.2",
"group_of": "video",
"description": "Wan 2.2\u2019s T2V mode transforms descriptive text prompts into high-quality, stylized video sequences. It excels at generating anime-style or cinematic visuals with smooth motion and strong thematic consistency.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "medium"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 5,
"minValue": 5,
"maxValue": 8,
"step": 3
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.2-text-to-video"
}
}
}
},
{
"name": "runway-act-two-i2v",
"category": "Image to Video",
"variant": "Act 2 Image to Video",
"family": "runway",
"group_of": "video",
"description": "Upload a single character image and a driving video \u2014 the model transfers facial expressions and head movements from the video onto your image, bringing it to life. It works with photos, illustrations, or stylized portraits, making them speak, blink, and move naturally. Ideal for avatars, AI presenters, digital actors, and story scenes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image. An image of your character. In the output, the character will use the reference video performance in its original static environment.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"reference_video_url": {
"description": "A video URL pointing to a video of a person performing in the manner that you would like your character to perform.",
"field": "video",
"type": "string",
"title": "Reference Video URL",
"name": "reference_video_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"image_url",
"reference_video_url"
],
"endpoint_url": "runway-act-two-i2v"
}
}
}
},
{
"name": "runway-act-two-v2v",
"category": "Video to Video",
"variant": "Act 2 Video to Video",
"family": "runway",
"group_of": "video",
"description": "Take an existing character video and sync it with the motion from a reference video. This lets you update facial expressions, head turns, and speech gestures while keeping the original look and style. It\u2019s perfect for reshooting performances, dubbing, or animating characters without re-rendering visuals.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "URL of the input video. An video of your character. In the output, the character will use the reference video performance in its original static environment.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"reference_video_url": {
"description": "A video URL pointing to a video of a person performing in the manner that you would like your character to perform.",
"field": "video",
"type": "string",
"title": "Reference Video URL",
"name": "reference_video_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"video_url",
"reference_video_url"
],
"endpoint_url": "runway-act-two-v2v"
}
}
}
},
{
"name": "pixverse-v4.5-i2v",
"category": "Image to Video",
"variant": "Image to Video",
"family": "pixverse-v4.5",
"group_of": "video",
"description": "Upload an image and PixVerse v4.5 will breathe life into it with smooth camera motion, realistic effects, and animated elements. Whether it\u2019s a portrait, landscape, or concept art, this mode turns still visuals into dynamic short videos.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide start frame and end frame images. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 2
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"360p",
"540p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds. 8s not supported for 1080p resolution.",
"default": 5,
"minValue": 5,
"maxValue": 8,
"step": 3
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "pixverse-v4.5-i2v"
}
}
}
},
{
"name": "pixverse-v4.5-t2v",
"category": "Text to Video",
"variant": "Text to Video",
"family": "pixverse-v4.5",
"group_of": "video",
"description": "PixVerse v4.5 transforms descriptive text into vivid, high-resolution video clips. It understands complex scenes, human motion, and cinematic camera angles \u2014 great for creative storytelling, trailers, and animated concepts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"360p",
"540p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds. 8s not supported for 1080p resolution.",
"default": 5,
"minValue": 5,
"maxValue": 8,
"step": 3
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "pixverse-v4.5-t2v"
}
}
}
},
{
"name": "vidu-v2.0-i2v",
"category": "Image to Video",
"variant": "Image to Video",
"family": "vidu-v2",
"group_of": "video",
"description": "Vidu's 2.0 model delivers advanced image-based video generation with enhanced lighting, emotion dynamics, and automatic frame interpolation for polished visual content.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide start frame and end frame images. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 2
},
"aspect_ratio": {
"enum": [
"16:9",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video. 16:9 for 360p/720p, 1:1 for 1080p are supported.",
"default": "16:9"
},
"resolution": {
"enum": [
"360p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"enum": [
4
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 4
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "vidu-v2.0-i2v"
}
}
}
},
{
"name": "vidu-v2.0-t2v",
"category": "Text to Video",
"variant": "Text to Video",
"family": "vidu-v2",
"group_of": "video",
"description": "Vidu's 2.0 model offers enhanced visual quality and comprehensive workflow support across multiple resolution options for versatile content creation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "9:16"
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
},
"duration": {
"enum": [
4
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 4
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "vidu-v2.0-t2v"
}
}
}
},
{
"name": "qwen-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "qwen",
"group_of": "image",
"description": "Generate high-quality, detailed images from text prompts in various styles \u2014 from realistic to artistic \u2014 perfect for creative visuals, product shots, and concept art.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"9:21",
"3:2",
"2:3"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "16:9"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "qwen-image"
}
}
}
},
{
"name": "runway-aleph-v2v",
"category": "Video to Video",
"variant": "Aleph Video to Video",
"family": "runway",
"group_of": "video",
"description": "Transform any input video into a new visual style or scene while preserving motion and structure. Aleph V2V lets you apply artistic looks, cinematic lighting, or thematic changes to existing footage.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "runway-aleph-v2v"
}
}
}
},
{
"name": "midjourney-v7-style-reference",
"category": "Image to Image",
"variant": "Style Reference",
"family": "midjourney",
"group_of": "image",
"description": "Generate images in the distinctive aesthetic of Midjourney v7 \u2014 blending cinematic depth, photorealism or painterly rendering, rich textures, and dynamic lighting. This style reference model helps you infuse any subject with the visual storytelling, composition, and high detail fidelity that Midjourney is known for. Ideal for concept art, stylized portraits, and stunning environment scenes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the image",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"speed": {
"enum": [
"relaxed",
"fast",
"turbo"
],
"title": "Speed",
"name": "speed",
"type": "string",
"description": "The speed of which corresponds to different speed of Midjourney",
"default": "relaxed"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"1:2",
"2:1",
"2:3",
"3:2",
"5:6",
"6:5"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"variety": {
"title": "Variety",
"name": "variety",
"type": "int",
"description": "Controls the diversity of generated images. Increment by 5 each time. Higher values create more diverse results. Lower values create more consistent results.",
"default": 5,
"minValue": 0,
"maxValue": 100,
"step": 5
},
"stylization": {
"title": "Stylization",
"name": "stylization",
"type": "int",
"description": "Controls the artistic style intensity. Higher values create more stylized results. Lower values create more realistic results.",
"default": 1,
"minValue": 0,
"maxValue": 1000,
"step": 1
},
"weirdness": {
"title": "Weirdness",
"name": "weirdness",
"type": "int",
"description": "Controls the creativity and uniqueness. Higher values create more unusual results. Lower values create more conventional results.",
"default": 1,
"minValue": 0,
"maxValue": 3000,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "midjourney-v7-style-reference"
}
}
}
},
{
"name": "midjourney-v7-omni-reference",
"category": "Image to Image",
"variant": "Omni Reference",
"family": "midjourney",
"group_of": "image",
"description": "Midjourney's Omni Reference lets you reuse characters, creatures, or styles from an existing image and place them into entirely new scenes. Simply provide a reference image (oref) and Midjourney will maintain identity, details, and visual consistency \u2014 ideal for storytelling, character design, or branding across multiple generations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the image",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"speed": {
"enum": [
"relaxed",
"fast",
"turbo"
],
"title": "Speed",
"name": "speed",
"type": "string",
"description": "The speed of which corresponds to different speed of Midjourney",
"default": "relaxed"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"1:2",
"2:1",
"2:3",
"3:2",
"5:6",
"6:5"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"weight": {
"title": "Weight",
"name": "weight",
"type": "int",
"description": "Weight allows you to control how much detail from your reference image appears in your new image.",
"default": 100,
"minValue": 1,
"maxValue": 1000,
"step": 1
},
"variety": {
"title": "Variety",
"name": "variety",
"type": "int",
"description": "Controls the diversity of generated images. Increment by 5 each time. Higher values create more diverse results. Lower values create more consistent results.",
"default": 5,
"minValue": 0,
"maxValue": 100,
"step": 5
},
"stylization": {
"title": "Stylization",
"name": "stylization",
"type": "int",
"description": "Controls the artistic style intensity. Higher values create more stylized results. Lower values create more realistic results.",
"default": 1,
"minValue": 0,
"maxValue": 1000,
"step": 1
},
"weirdness": {
"title": "Weirdness",
"name": "weirdness",
"type": "int",
"description": "Controls the creativity and uniqueness. Higher values create more unusual results. Lower values create more conventional results.",
"default": 1,
"minValue": 0,
"maxValue": 3000,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "midjourney-v7-omni-reference"
}
}
}
},
{
"name": "minimax-image-01-subject-reference",
"category": "Image to Image",
"variant": "Subject Reference",
"family": "minimax",
"group_of": "image",
"description": "Minimax\u2019s I2I \u201cSubject Reference\u201d model enables you to transform images while preserving the appearance of a subject using a single reference image. Ideal for maintaining character likeness\u2014features, clothing, or expression\u2014across different styles or settings.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image (max 1500 characters).",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"3:2",
"2:3",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "minimax-01-subject-reference"
}
}
}
},
{
"name": "ideogram-character",
"category": "Image to Image",
"variant": "Character",
"family": "ideogram",
"group_of": "image",
"description": "Ideogram\u2019s Character Reference model enables consistent character generation using just one reference image. Upload a clear character portrait\u2014and you can place that character in unlimited scenes, styles, poses, or narratives with visual fidelity maintained across all outputs.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image (max 1500 characters).",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"render_speed": {
"enum": [
"Turbo",
"Balanced",
"Quality"
],
"title": "Render Speed",
"name": "render_speed",
"type": "string",
"description": "The rendering speed to use.",
"default": "Balanced"
},
"style": {
"enum": [
"Auto",
"Realistic",
"Fiction"
],
"title": "Style",
"name": "style",
"type": "string",
"description": "The style type to generate with.",
"default": "Auto"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "ideogram-character"
}
}
}
},
{
"name": "flux-pulid",
"category": "Image to Image",
"variant": "Pulid Image to Image",
"family": "flux",
"group_of": "image",
"description": "Flux PuLID is an innovative image-to-image model that enables consistent face rendering across different styles or scenes\u2014without needing any model fine-tuning. By providing a reference image (e.g., a portrait), the model generates new visuals while maintaining your subject\u2019s identity with high fidelity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image (max 1500 characters).",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "flux-pulid"
}
}
}
},
{
"name": "sync-lipsync",
"category": "Audio to Video",
"variant": "sync",
"family": "lipsync",
"group_of": "ai-tools",
"description": "Generate realistic lipsync animations from audio using advanced algorithms for high-quality synchronization.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"audio_url",
"video_url"
],
"endpoint_url": "sync-lipsync"
}
}
}
},
{
"name": "latent-sync",
"category": "Audio to Video",
"variant": "latent",
"family": "lipsync",
"group_of": "ai-tools",
"description": "LatentSync is a video-to-video model that generates lip sync animations from audio using advanced algorithms for high-quality synchronization.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"audio_url",
"video_url"
],
"endpoint_url": "latentsync-video"
}
}
}
},
{
"name": "creatify-lipsync",
"category": "Audio to Video",
"variant": "Creatify",
"family": "lipsync",
"group_of": "ai-tools",
"description": "Realistic lipsync video - optimized for speed, quality, and consistency.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"audio_url",
"video_url"
],
"endpoint_url": "creatify-lipsync"
}
}
}
},
{
"name": "veed-lipsync",
"category": "Audio to Video",
"variant": "Veed",
"family": "lipsync",
"group_of": "ai-tools",
"description": "Generate realistic lipsync from any audio using VEED's latest model",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"audio_url",
"video_url"
],
"endpoint_url": "veed-lipsync"
}
}
}
},
{
"name": "luma-modify-video",
"category": "Video to Video",
"variant": "Modify V2V",
"family": "luma",
"group_of": "video",
"description": "Luma Modify Video lets you transform an existing video into a new creative scene while keeping the original motion and timing intact. The result is a new video with the same movements but a completely fresh look, atmosphere, or theme.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video. Max size 500MB. Max duration 10 seconds. Supported formats MP4, MOV, AVI.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "luma-modify-video"
}
}
}
},
{
"name": "luma-flash-reframe",
"category": "Video to Video",
"variant": "Flash Reframe V2V",
"family": "luma",
"group_of": "video",
"description": "Transform and resize your videos effortlessly with Ray 2 Flash Reframe. This tool intelligently expands or adjusts your video\u2019s aspect ratio\u2014adding visually consistent content to the sides, top, or bottom\u2014without altering the original subject.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Optional prompt for reframing.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video to reframe.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"image_url": {
"description": "Optional URL of the first frame image for reframing.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"enum": [
5,
8,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 5
}
},
"title": "BaseInput",
"required": [
"video_url",
"aspect_ratio"
],
"endpoint_url": "luma-flash-reframe"
}
}
}
},
{
"name": "qwen-image-edit",
"category": "Image to Image",
"variant": "Edit Image",
"family": "qwen",
"group_of": "image",
"description": "The Qwen Edit Image Model allows you to modify existing images using text-based editing prompts. Instead of generating from scratch, you can upload a base image and describe the desired changes (e.g., replacing objects, altering colors, adding new elements).",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image to used to transform the image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9",
"9:21",
"3:2",
"2:3"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "qwen-image-edit"
}
}
}
},
{
"name": "vidu-q1-reference",
"category": "Image to Video",
"variant": "Reference I2V",
"family": "vidu-q1",
"group_of": "video",
"description": "Vidu Q1 enables you to generate cinematic 1080p videos using multiple visual references\u2014up to seven images\u2014and text prompts. Designed for consistency, it preserves character appearance, props, and backgrounds across scenes while adding new motion and narrative elements.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the desired video content.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for create consistent character video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 7
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "vidu-q1-reference"
}
}
}
},
{
"name": "wan2.2-5b-fast-t2v",
"category": "Text to Video",
"variant": "Fast Text to Video",
"family": "wan2.2",
"group_of": "video",
"description": "Wan 2.2 Fast is a lightweight, high-speed version of the Wan 2.2 model, optimized for quick text-to-video generation. It trades some cinematic detail for rapid results, making it perfect for prototyping, previews, social media clips, and quick storytelling.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"580p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.2-5b-fast-t2v"
}
}
}
},
{
"name": "minimax-hailuo-02-standard-i2v",
"category": "Image to Video",
"variant": "Standard I2V",
"family": "minimax-2",
"group_of": "video",
"description": "Transforms an image into video with light, natural motion. Great for social media, quick animations, and previews.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"end_image_url": {
"description": "URL of the end image url.",
"field": "image",
"type": "string",
"title": "End Image URL",
"name": "end_image_url"
},
"duration": {
"enum": [
6,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"resolution": {
"enum": [
"512P",
"768P"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "512P"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "minimax-hailuo-02-standard-i2v"
}
}
}
},
{
"name": "minimax-hailuo-02-standard-t2v",
"category": "Text to Video",
"variant": "Standard T2V",
"family": "minimax-2",
"group_of": "video",
"description": "Fast and lightweight text-to-video generation. Ideal for quick drafts, previews, or playful content where speed matters more than cinematic quality.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"duration": {
"enum": [
6,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"resolution": {
"enum": [
"768P"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "768P"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "minimax-hailuo-02-standard-t2v"
}
}
}
},
{
"name": "minimax-hailuo-02-pro-i2v",
"category": "Image to Video",
"variant": "Pro I2V",
"family": "minimax-2",
"group_of": "video",
"description": "Advanced image-to-video with cinematic realism. Adds dynamic camera motion, realistic physics, and atmospheric detail for storytelling.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"end_image_url": {
"description": "URL of the end image url.",
"field": "image",
"type": "string",
"title": "End Image URL",
"name": "end_image_url"
},
"duration": {
"enum": [
6
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "minimax-hailuo-02-pro-i2v"
}
}
}
},
{
"name": "minimax-hailuo-02-pro-t2v",
"category": "Text to Video",
"variant": "Pro T2V",
"family": "minimax-2",
"group_of": "video",
"description": "High-fidelity text-to-video with cinematic rendering. Best for storytelling, cinematic clips, or realistic visuals with depth, atmosphere, and detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"duration": {
"enum": [
6
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"resolution": {
"enum": [
"1080P"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080P"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "minimax-hailuo-02-pro-t2v"
}
}
}
},
{
"name": "ai-dance-effects",
"category": "Video to Video",
"variant": "AI Dance Effects",
"family": "effects",
"group_of": "video",
"description": "Bring your characters and worlds to life with AI Dance Effects \u2014 a creative video effect that adds playful, dynamic, and cinematic motion to your generations. AI Dance Effects lets you guide how characters move, react, and express themselves.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Optional prompt for generating video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"image_url",
"video_url"
],
"endpoint_url": "ai-dance-effects"
}
}
}
},
{
"name": "video-effects",
"category": "Image to Video",
"variant": "Video Effects",
"family": "effects",
"group_of": "ai-effects",
"description": "AI Video Effects applies advanced visual transformations, color grading, and cinematic filters to create stunning videos from images.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"name": {
"enum": [
"Balloon Flyaway",
"Blow Kiss",
"Body Shake",
"Break Glass",
"Carry Me",
"Cartoon Doll",
"Cheek Kiss",
"Child Memory",
"Couple Arrival",
"Fairy Me",
"Fashion Stride",
"Fisherman",
"Flower Receive",
"Flying",
"French Kiss",
"Gender Swap",
"Golden Epoch",
"Hair Swap",
"Hugging",
"Jiggle Up",
"Kissing Pro",
"Live Memory",
"Love Drop",
"Melt",
"Minecraft",
"Muscling",
"Nap Me 360p",
"Paperman",
"Pilot",
"Pinch",
"Pixel Me",
"Romantic Lift",
"Sexy Me",
"Slice Therapy",
"Soul Depart",
"Split Stance Human",
"Squid Game",
"Toy Me",
"Walk Forward",
"Zoom In Fast",
"Zoom Out"
],
"description": "The type of effect to apply to the video.",
"type": "string",
"title": "Effect Name",
"name": "name",
"default": "Balloon Flyaway"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "video-effects"
},
"examples": [
{
"name": "Balloon Flyaway",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/balloon-flyaway.mp4"
},
{
"name": "Blow Kiss",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/blow-kiss.mp4"
},
{
"name": "Body Shake",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/body-shake.mp4"
},
{
"name": "Break Glass",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/break-glass.mp4"
},
{
"name": "Carry Me",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/carry-me.mp4"
},
{
"name": "Cartoon Doll",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/cartoon-doll.mp4"
},
{
"name": "Cheek Kiss",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/cheek-kiss.mp4"
},
{
"name": "Child Memory",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/child-memory.mp4"
},
{
"name": "Couple Arrival",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/couple-arrival.mp4"
},
{
"name": "Fairy Me",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/fairy-me.mp4"
},
{
"name": "Fashion Stride",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/fashion-stride.mp4"
},
{
"name": "Fisherman",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/fisherman.mp4"
},
{
"name": "Flower Receive",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/flower-receive.mp4"
},
{
"name": "Flying",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/flying.mp4"
},
{
"name": "French Kiss",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/french-kiss.mp4"
},
{
"name": "Gender Swap",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/gender-swap.mp4"
},
{
"name": "Golden Epoch",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/golden-epoch.mp4"
},
{
"name": "Hair Swap",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/hair-swap.mp4"
},
{
"name": "Hugging",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/hugging.mp4"
},
{
"name": "Jiggle Up",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/jiggle-up.mp4"
},
{
"name": "Kissing Pro",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/kissing-pro.mp4"
},
{
"name": "Live Memory",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/live-memory.mp4"
},
{
"name": "Love Drop",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/love-drop.mp4"
},
{
"name": "Melt",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/melt.mp4"
},
{
"name": "Minecraft",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/minecraft.mp4"
},
{
"name": "Muscling",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/muscling.mp4"
},
{
"name": "Nap Me 360p",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/nap-me-360p.mp4"
},
{
"name": "Paperman",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/paperman.mp4"
},
{
"name": "Pilot",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/pilot.mp4"
},
{
"name": "Pinch",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/pinch.mp4"
},
{
"name": "Pixel Me",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/pixel-me.mp4"
},
{
"name": "Romantic Lift",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/romantic-lift.mp4"
},
{
"name": "Sexy Me",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/sexy-me.mp4"
},
{
"name": "Slice Therapy",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/slice-therapy.mp4"
},
{
"name": "Soul Depart",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/soul-depart.mp4"
},
{
"name": "Split Stance Human",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/split-stance-human.mp4"
},
{
"name": "Squid Game",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/squid-game.mp4"
},
{
"name": "Toy Me",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/toy-me.mp4"
},
{
"name": "Walk Forward",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/walk-forward.mp4"
},
{
"name": "Zoom In Fast",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/zoom-in-fast.mp4"
},
{
"name": "Zoom Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/zoom-out.mp4"
}
]
}
}
},
{
"name": "image-effects",
"category": "Image to Image",
"variant": "Image Effects",
"family": "effects",
"group_of": "ai-effects",
"description": "AI Image Effects applies advanced visual transformations, color grading, and cinematic filters to create stunning images from a image.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"name": {
"enum": [
"Acryclic Ornaments",
"Advanced Photography",
"American Comic Style",
"Angel Figurine",
"Blurry Selfie",
"Cyberpunk",
"Exotic Charm",
"Felt 3D Polaroid",
"Felt Keychain",
"Furry Dream Doll",
"Futuristic American Comics",
"Glass Ball",
"In The Stadium",
"Lofi Pixel Character",
"Lying On Fluffy Belly",
"Landscape Mini World",
"My World",
"Plastic Bubble Figure"
],
"description": "The type of effect to apply to the image.",
"type": "string",
"title": "Effect Name",
"name": "name",
"default": "Angel Figurine"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "image-effects"
},
"examples": [
{
"name": "Acryclic Ornaments",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/acrylic-ornaments.jpg"
},
{
"name": "Advanced Photography",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/advanced-photography.jpg"
},
{
"name": "American Comic Style",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/american-comic-style.jpg"
},
{
"name": "Angel Figurine",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/angel-figurine.jpg"
},
{
"name": "Blurry Selfie",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/blurry-selfie-luffy.jpg"
},
{
"name": "Cyberpunk",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/cyberpunk.jpg"
},
{
"name": "Exotic Charm",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/exotic-charm.jpg"
},
{
"name": "Felt 3D Polaroid",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/felt-3d-polaroid.jpg"
},
{
"name": "Felt Keychain",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/felt-keychain.jpg"
},
{
"name": "Furry Dream Doll",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/furry-dream-doll.jpg"
},
{
"name": "Futuristic American Comics",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/futuristic-american-comics.jpg"
},
{
"name": "Glass Ball",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/glass-ball.jpg"
},
{
"name": "In The Stadium",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/in-the-stadium.jpg"
},
{
"name": "Lofi Pixel Character",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/lofi-pixel-character-mini-card.jpg"
},
{
"name": "Lying On Fluffy Belly",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/lying-in-fluffy-belly.jpg"
},
{
"name": "Landscape Mini World",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/micro-landscape-mini-world.jpg"
},
{
"name": "My World",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/my-world.jpg"
},
{
"name": "Plastic Bubble Figure",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/ai_effects/plastic-bubble-figure.jpg"
}
]
}
}
},
{
"name": "seedance-lite-i2v",
"category": "Image to Video",
"variant": "Lite Image to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Lite I2V version animates static images into short videos quickly, focusing on basic motion effects and efficient processing\u2014best suited for fast demos or mobile-friendly use.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 12,
"step": 1
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "seedance-lite-i2v"
}
}
}
},
{
"name": "seedance-lite-t2v",
"category": "Text to Video",
"variant": "Lite Text to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Lite T2V offers quick video generation from text with decent visual quality and motion. Ideal for fast previews, prototyping, or lightweight use cases where speed matters more than fine detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 12,
"step": 1
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-lite-t2v"
}
}
}
},
{
"name": "seedance-pro-i2v",
"category": "Image to Video",
"variant": "Pro Image to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Pro I2V advanced model animates still images into stunning short videos, preserving intricate visual details and applying smooth motion dynamics, ideal for high-end visuals and cinematic edits.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 12,
"step": 1
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "seedance-pro-i2v"
}
}
}
},
{
"name": "seedance-pro-t2v",
"category": "Text to Video",
"variant": "Pro Text to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Pro delivers high-fidelity video generation from text, producing rich visuals, smooth camera movement, and realistic scenes. Best for storytelling, content creation, and visual production.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 12,
"step": 1
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-pro-t2v"
}
}
}
},
{
"name": "sdxl-lora",
"category": "Training",
"variant": "LoRA",
"family": "sdxl",
"group_of": "image",
"description": "The SDXL LoRA image model enhances Stable Diffusion XL with specialized fine-tuning, letting you generate images in unique styles, characters, or themes. By applying LoRA weights, you can create visuals that match a specific aesthetic, celebrity look, anime style, or custom-trained subject.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"lora_list": {
"title": "LoRA List",
"name": "lora_list",
"type": "array",
"items": {
"type": "object",
"properties": {
"path": {
"type": "string",
"format": "url",
"title": "Path",
"name": "path",
"description": "URL or the path to the LoRA weights."
},
"scale": {
"type": "number",
"title": "Scale",
"name": "scale",
"description": "The scale of the LoRA weight. This is used to scale the LoRA weight before merging it with the base model. Default value: 1",
"minValue": 0,
"maxValue": 4,
"step": 0.01,
"default": 1
}
}
},
"description": "The LoRA weights for generating video",
"maxItems": 4
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1,
"isEdit": true
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "sdxl-lora-image"
}
}
}
},
{
"name": "ideogram-v3-t2i",
"category": "Text to Image",
"variant": "v3 Text to Image",
"family": "ideogram",
"group_of": "image",
"description": "Ideogram v3 is an advanced text-to-image model designed for creating highly detailed and visually striking images directly from text prompts. It\u2019s especially good for artistic compositions, design mockups, concept art, and photorealistic scenes. With strong support for text rendering inside images, it\u2019s widely used for posters, typography-based art, and creative branding.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"render_speed": {
"enum": [
"Turbo",
"Balanced",
"Quality"
],
"title": "Render Speed",
"name": "render_speed",
"type": "string",
"description": "The rendering speed to use.",
"default": "Balanced"
},
"style": {
"enum": [
"Auto",
"General",
"Realistic",
"Design"
],
"title": "Style",
"name": "style",
"type": "string",
"description": "The style type to generate with.",
"default": "Auto"
},
"aspect_ratio": {
"enum": [
"1:1",
"3:4",
"4:3",
"9:16",
"16:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "ideogram-v3-t2i"
}
}
}
},
{
"name": "nano-banana-edit",
"category": "Image to Image",
"variant": "Edit Image",
"family": "nano",
"group_of": "image",
"description": "Nano Banana is a mysterious, high-performance image model. It excels at precise, language-driven edits and consistent character preservation, allowing users to modify images with natural text commands.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "List of URLs of input images for editing.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 10
},
"aspect_ratio": {
"enum": [
"Auto",
"1:1",
"3:4",
"4:3",
"9:16",
"16:9",
"3:2",
"2:3",
"5:4",
"4:5",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "Auto"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "nano-banana-edit"
}
}
}
},
{
"name": "nano-banana",
"category": "Text to Image",
"variant": "Text to Image",
"family": "nano",
"group_of": "image",
"description": "Nano Banana is an advanced AI model excelling in natural language-driven image generation and editing. It produces hyper-realistic, physics-aware visuals with seamless style transformations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"3:4",
"4:3",
"9:16",
"16:9",
"3:2",
"2:3",
"5:4",
"4:5",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "nano-banana"
}
}
}
},
{
"name": "pixverse-v5-i2v",
"category": "Image to Video",
"variant": "Image to Video",
"family": "pixverse-v5",
"group_of": "video",
"description": "PixVerse V5 delivers a major leap forward in AI-powered video creation \u2014 now featuring smoother motion, ultra-high resolution, and expanded visual effects.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide start frame and end frame images. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 2
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"360p",
"540p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 8,
"step": 3
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "pixverse-v5-i2v"
}
}
}
},
{
"name": "pixverse-v5-t2v",
"category": "Text to Video",
"variant": "Text to Video",
"family": "pixverse-v5",
"group_of": "video",
"description": "PixVerse V5 delivers a major leap forward in AI-powered video creation \u2014 now featuring smoother motion, ultra-high resolution, and expanded visual effects.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"360p",
"540p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 8,
"step": 3
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "pixverse-v5-t2v"
}
}
}
},
{
"name": "wan2.2-speech-to-video",
"category": "Audio to Video",
"variant": "Audio to Video",
"family": "wan2.2",
"group_of": "video",
"description": "WAN2.2 Speech-to-Video transforms a static image into a talking video by synchronizing lip movements and facial expressions with an audio input. Simply provide a character image along with a speech dialogue, and the model generates a natural, expressive video where the subject speaks your lines.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"image_url",
"audio_url"
],
"endpoint_url": "wan2.2-speech-to-video"
}
}
}
},
{
"name": "google-imagen4",
"category": "Text to Image",
"variant": "Imagen 4",
"family": "google",
"group_of": "image",
"description": "Google Imagen 4 is the latest text-to-image AI model from DeepMind, designed to produce stunningly photorealistic images with crisp detail, accurate text rendering, and creative flexibility. It supports high-resolution output (up to 2K), generates visuals in seconds, and embeds SynthID watermarks for authenticity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "google-imagen4"
}
}
}
},
{
"name": "google-imagen4-fast",
"category": "Text to Image",
"variant": "Imagen 4 Fast",
"family": "google",
"group_of": "image",
"description": "Imagen 4 Fast is optimized for speed and accessibility, allowing you to generate high-quality images in seconds. While slightly less detailed than the Ultra version, it excels at rapid ideation, drafts, storyboarding, and casual creativity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1,
"isEdit": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "google-imagen4-fast"
}
}
}
},
{
"name": "google-imagen4-ultra",
"category": "Text to Image",
"variant": "Imagen 4 Ultra",
"family": "google",
"group_of": "image",
"description": "Imagen 4 Ultra is Google\u2019s flagship model, designed for photorealism, rich textures, and production-level imagery. It produces crisp, high-resolution visuals with advanced detail, lighting precision, and natural compositions.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "google-imagen4-ultra"
}
}
}
},
{
"name": "seedance-lite-reference-video",
"category": "Image to Video",
"variant": "Lite Reference to Video",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Lite's Reference-to-Video feature allows you to supply up to 4 images as reference inputs. The model intelligently blends aspects from these images to generate a cohesive, high-quality video.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 4
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 12,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "seedance-lite-reference-to-video"
}
}
}
},
{
"name": "wan2.1-reference-video",
"category": "Image to Video",
"variant": "Reference to Video",
"family": "wan2.1",
"group_of": "video",
"description": "WAN 2.1 is an advanced AI model that transforms one or more reference images into a coherent, animated video. By combining characters, objects, or environments from multiple images, it creates smooth motion sequences while preserving realism, style, and fine details.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 5
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "wan2.1-reference-video"
}
}
}
},
{
"name": "infinitetalk-image-to-video",
"category": "Audio to Video",
"variant": "Image to Video",
"family": "infinite-talk",
"group_of": "video",
"description": "InfiniteTalk Image-to-Video brings still portraits and character photos to life by generating natural, realistic talking videos. You provide a single face image and a dialogue script, and the model animates lip movement, facial expressions, and subtle head gestures to match the speech.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"image_url",
"audio_url"
],
"endpoint_url": "infinitetalk-image-to-video"
}
}
}
},
{
"name": "infinitetalk-video-to-video",
"category": "Video to Video",
"variant": "Audio to Video",
"family": "infinite-talk",
"group_of": "video",
"description": "InfiniteTalk Video-to-Video enhances or transforms existing videos by syncing the subject\u2019s lip movements and facial expressions with new dialogue or speech. Instead of starting from a still image, you provide a video clip, and the model seamlessly reanimates the speaker\u2019s mouth and expressions to match the script.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"video_url",
"audio_url"
],
"endpoint_url": "infinitetalk-video-to-video"
}
}
}
},
{
"name": "ideogram-v3-reframe",
"category": "Image to Image",
"variant": "v3 Reframe",
"family": "ideogram",
"group_of": "image",
"description": "Ideogram V3 Reframe is a specialized image-to-image model built on Ideogram 3.0, designed to intelligently extend and adapt images across diverse aspect ratios and resolutions. Leveraging advanced AI outpainting, it preserves visual consistency while enabling creative reframing for digital, print, and video content.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image used to generate image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"render_speed": {
"enum": [
"Turbo",
"Balanced",
"Quality"
],
"title": "Render Speed",
"name": "render_speed",
"type": "string",
"description": "The rendering speed to use.",
"default": "Balanced"
},
"style": {
"enum": [
"Auto",
"General",
"Realistic",
"Design"
],
"title": "Style",
"name": "style",
"type": "string",
"description": "The style type to generate with.",
"default": "Auto"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "ideogram-v3-reframe"
}
}
}
},
{
"name": "sdxl-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "sdxl",
"group_of": "image",
"description": "SDXL is a high-quality, large Stable Diffusion model for creating photorealistic and stylized images from text. It excels at fine detail, realistic lighting, and complex scenes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "sdxl-image"
}
}
}
},
{
"name": "bytedance-seedream-v4",
"category": "Text to Image",
"variant": "Text to Image v4",
"family": "seedream",
"group_of": "image",
"description": "Seedream v4 generates stunning, high-fidelity images from text prompts. It\u2019s designed for creativity with strong support for realism, fantasy, and artistic styles.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1K",
"2K",
"4K"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "Resolution of the output image.",
"default": "4K"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "bytedance-seedream-v4"
}
}
}
},
{
"name": "bytedance-seedream-edit-v4",
"category": "Image to Image",
"variant": "Edit Image v4",
"family": "seedream",
"group_of": "image",
"description": "Seedream v4 Edit refines or transforms existing images based on a new prompt and a reference. Instead of masking, you provide a source image and describe how it should be altered \u2014 adjusting style, details, or replacing elements while keeping the subject consistent.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 10
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1K",
"2K",
"4K"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "Resolution of the output image.",
"default": "4K"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "bytedance-seedream-edit-v4"
}
}
}
},
{
"name": "hunyuan-image-2.1",
"category": "Text to Image",
"variant": "Text to Image v2.1",
"family": "hunyuan",
"group_of": "image",
"description": "Hunyuan Image is a powerful text-to-image generation model that produces photorealistic and highly detailed visuals. It excels at creating portraits, environments, and concept art with strong consistency and realism. Designed for versatility, it supports both natural photography styles and imaginative artistic outputs.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "hunyuan-image-2.1"
}
}
}
},
{
"name": "chroma-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "chroma",
"group_of": "image",
"description": "Croma Image is an advanced text-to-image generation model designed for high-quality, creative, and versatile visuals. It can produce anything from photorealistic portraits and products to imaginative concept art, fantasy illustrations, and cinematic scenes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "chroma-image"
}
}
}
},
{
"name": "nano-banana-effects",
"category": "Image to Image",
"variant": "Image Effects",
"family": "nano",
"group_of": "ai-effects",
"description": "Nano Banana Effects is a creative visual effects model designed to transform ordinary images into fun, stylized, and eye-catching results. It applies artistic filters, 3D styles, cartoon transformations, and trending viral looks with a single click.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"name": {
"enum": [
"3D Figurine",
"16bit Game Character",
"1920s Decade",
"1950s Decade",
"1970s Decade",
"1980s Decade",
"Action Figure",
"American Gothic Art",
"Egypts Landmark",
"Eiffel Tower Landmark",
"Famous Art",
"Great Wall of China Landmark",
"Mona Lisa Art",
"Persistent Memory Art",
"Statue of Liberty Landmark",
"Taj Mahal Landmark",
"Vincent Van Gogh Art"
],
"description": "The type of effect to apply to the image.",
"type": "string",
"title": "Effect Name",
"name": "name",
"default": "3D Figurine"
},
"aspect_ratio": {
"enum": [
"Auto",
"1:1",
"3:4",
"4:3",
"9:16",
"16:9",
"3:2",
"2:3",
"5:4",
"4:5",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "Auto"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "nano-banana-effects"
},
"examples": [
{
"name": "3D Figurine",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/3d-figurine.jpg"
},
{
"name": "16bit Game Character",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/16bit-game-character.jpg"
},
{
"name": "1920s Decade",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/1920s-decade.jpg"
},
{
"name": "1950s Decade",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/1950s-decade.jpg"
},
{
"name": "1970s Decade",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/1970s-decade.jpg"
},
{
"name": "1980s Decade",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/1980s-decade.jpg"
},
{
"name": "Action Figure",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/action-figure.jpg"
},
{
"name": "American Gothic Art",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/american-gothic-art.jpg"
},
{
"name": "Egypts Landmark",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/egypts-landmark.jpg"
},
{
"name": "Eiffel Tower Landmark",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/eiffel-tower-landmark.jpg"
},
{
"name": "Famous Art",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/famous-art.jpg"
},
{
"name": "Great Wall of China Landmark",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/great-wall-of-china-landmark.jpg"
},
{
"name": "Mona Lisa Art",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/mona-lisa-art.jpg"
},
{
"name": "Persistent Memory Art",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/persistent-memory-art.jpg"
},
{
"name": "Statue of Liberty Landmark",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/statue-of-liberty-landmark.jpg"
},
{
"name": "Taj Mahal Landmark",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/taj-mahal-landmark.jpg"
},
{
"name": "Vincent Van Gogh Art",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/vincent-van-gogh-art.jpg"
}
]
}
}
},
{
"name": "flux-kontext-effects",
"category": "Image to Image",
"variant": "Image Effects",
"family": "kontext",
"group_of": "ai-effects",
"description": "Flux Kontext Effects is a creative image and video model that applies stylized transformations, cinematic filters, and artistic reinterpretations to your inputs. Instead of generating new content from scratch, it enhances or reimagines existing images and videos with unique looks \u2014 ranging from surreal effects to realistic cinematic moods.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"name": {
"enum": [
"Age Progression",
"Background Change",
"Cartoonify",
"Color Correction",
"Expression Change",
"Face Enhancement",
"Hair Change",
"Object Removal",
"Professional Photo",
"Scene Composition",
"Style Transfer",
"Time of Day",
"Weather Effect"
],
"description": "The type of effect to apply to the image.",
"type": "string",
"title": "Effect Name",
"name": "name",
"default": "Age Progression"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "flux-kontext-effects"
},
"examples": [
{
"name": "Age Progression",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/age_progression.jpg"
},
{
"name": "Background Change",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/background_change.jpg"
},
{
"name": "Cartoonify",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/cartoonify.jpg"
},
{
"name": "Color Correction",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/color_correction.jpg"
},
{
"name": "Expression Change",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/expression_change.jpg"
},
{
"name": "Face Enhancement",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/face_enhancement.jpg"
},
{
"name": "Hair Change",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/hair_change.jpg"
},
{
"name": "Object Removal",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/object_removal.jpg"
},
{
"name": "Professional Photo",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/professional_photo.jpg"
},
{
"name": "Scene Composition",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/scene_composition.jpg"
},
{
"name": "Style Transfer",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/style_transfer.jpg"
},
{
"name": "Time of Day",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/time_of_day.jpg"
},
{
"name": "Weather Effect",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/kontext_effect/weather_effect.jpg"
}
]
}
}
},
{
"name": "ai-video-upscaler",
"category": "Video to Video",
"variant": "Video Upscaler",
"family": "tools",
"group_of": "ai-tools",
"description": "The AI Video Upscaler is a powerful tool designed to enhance the resolution and quality of videos. Whether you're working with low-resolution videos that need a boost or aiming to improve the clarity of existing footage, this upscaler leverages advanced machine learning models to deliver high-quality, upscaled videos.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"resolution": {
"enum": [
"720p",
"1080p",
"2k",
"4k"
],
"description": "Target resolution to upscale.",
"type": "string",
"title": "Resolution",
"name": "resolution",
"default": "720p"
},
"copy_audio": {
"type": "boolean",
"title": "Copy Audio",
"name": "copy_audio",
"description": "Whether to copy the original video's audio to the upscaled video.",
"default": true
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": "ai-video-upscaler"
}
}
}
},
{
"name": "flux-redux",
"category": "Image to Image",
"variant": "Redux Image to Image",
"family": "flux",
"group_of": "image",
"description": "Flux Redux is a transformation model that reimagines or enhances your input images while preserving their main structure and subject. It\u2019s built for creative refinement \u2014 whether you want style transfer, artistic reinterpretation, cinematic polish, or mood transformation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image (max 1500 characters).",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"3:2",
"2:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "flux-redux"
}
}
}
},
{
"name": "flux-krea-dev",
"category": "Text to Image",
"variant": "Krea Dev",
"family": "flux",
"group_of": "image",
"description": "Flux Krea Dev is a text-to-image model built by Black Forest Labs in collaboration with Krea AI, designed to generate highly photorealistic images that avoid the common 'AI look' artifacts (plastic skin, overexposed lighting, synthetic textures). It emphasizes real texture, natural lighting, and aesthetic control.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image. The length of the prompt must be between 2 and 3000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"3:2",
"2:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 4,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-krea-dev"
}
}
}
},
{
"name": "perfect-pony-xl",
"category": "Text to Image",
"variant": "Text to Image",
"family": "pony",
"group_of": "image",
"description": "Pony XL is a high-quality image generation model based on Stable Diffusion XL architecture. It specializes in character art, hybrid styles, and producing detailed, polished visuals even with simpler prompts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "perfect-pony-xl"
}
}
}
},
{
"name": "neta-lumina",
"category": "Text to Image",
"variant": "Text to Image",
"family": "neta",
"group_of": "image",
"description": "Neta Lumina is a powerful anime-style text-to-image model developed by Neta.art Lab. It\u2019s built on Lumina-Image-2.0, fine-tuned with over 13 million high-quality anime images. It offers strong understanding of multilingual prompts, excellent detail fidelity, support for Danbooru tags, and leaning into niche styles like furry, Guofeng, pets, scenic backgrounds, etc.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "neta-lumina"
}
}
}
},
{
"name": "wan2.2-edit-video",
"category": "Video to Video",
"variant": "Edit Video",
"family": "wan2.2",
"group_of": "video",
"description": "Easily modify existing videos using simple text commands. With Wan 2.2 Video-Edit, you can change attire, character appearance, or other visual elements directly within your video\u2014no need to start from scratch. Works on uploads of 480p or 720p, for up to two minutes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "wan2.2-edit-video"
}
}
}
},
{
"name": "kling-v1-avatar-standard",
"category": "Audio to Video",
"variant": "Standard A2V",
"family": "kling-v1",
"group_of": "video",
"description": "Kling AI Avatar Standard creates talking avatar videos from a single image + audio input. It supports realistic humans, animals, or stylized characters, producing lip-synced avatar videos easily.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
}
},
"title": "BaseInput",
"required": [
"image_url",
"audio_url"
],
"endpoint_url": "kling-v1-avatar-standard"
}
}
}
},
{
"name": "kling-v1-avatar-pro",
"category": "Audio to Video",
"variant": "Pro A2V",
"family": "kling-v1",
"group_of": "video",
"description": "Kling AI Avatar Pro is the premium tier for making high-quality talking avatars. You upload a character image plus an audio file, and the model generates a realistic avatar video with lip-sync.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
}
},
"title": "BaseInput",
"required": [
"image_url",
"audio_url"
],
"endpoint_url": "kling-v1-avatar-pro"
}
}
}
},
{
"name": "heygen-video-translate",
"category": "Video to Video",
"variant": "Video Translate",
"family": "tools",
"group_of": "video",
"description": "Convert any video into 175+ languages with synchronized voice translation, AI-voice cloning, and accurate lip sync. Just upload your video (or provide a link), select a target language, and HeyGen recreates the speech in that language. 0.05$ per second.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"language": {
"enum": [
"English",
"Spanish",
"French",
"Hindi",
"Italian",
"German",
"Polish",
"Portuguese",
"Chinese",
"Japanese",
"Dutch",
"Turkish",
"Korean",
"Danish",
"Arabic",
"Romanian",
"Mandarin",
"Filipino",
"Swedish",
"Indonesian",
"Ukrainian",
"Greek",
"Czech",
"Bulgarian",
"Malay",
"Slovak",
"Croatian",
"Tamil",
"Finnish",
"Russian",
"Afrikaans (South Africa)",
"Albanian (Albania)",
"Amharic (Ethiopia)",
"Arabic (Algeria)",
"Arabic (Bahrain)",
"Arabic (Egypt)",
"Arabic (Iraq)",
"Arabic (Jordan)",
"Arabic (Kuwait)",
"Arabic (Lebanon)",
"Arabic (Libya)",
"Arabic (Morocco)",
"Arabic (Oman)",
"Arabic (Qatar)",
"Arabic (Saudi Arabia)",
"Arabic (Syria)",
"Arabic (Tunisia)",
"Arabic (United Arab Emirates)",
"Arabic (Yemen)",
"Armenian (Armenia)",
"Azerbaijani (Latin, Azerbaijan)",
"Bangla (Bangladesh)",
"Basque",
"Bengali (India)",
"Bosnian (Bosnia and Herzegovina)",
"Bulgarian (Bulgaria)",
"Burmese (Myanmar)",
"Catalan",
"Chinese (Cantonese, Traditional)",
"Chinese (Jilu Mandarin, Simplified)",
"Chinese (Mandarin, Simplified)",
"Chinese (Northeastern Mandarin, Simplified)",
"Chinese (Southwestern Mandarin, Simplified)",
"Chinese (Taiwanese Mandarin, Traditional)",
"Chinese (Wu, Simplified)",
"Chinese (Zhongyuan Mandarin Henan, Simplified)",
"Chinese (Zhongyuan Mandarin Shaanxi, Simplified)",
"Croatian (Croatia)",
"Czech (Czechia)",
"Danish (Denmark)",
"Dutch (Belgium)",
"Dutch (Netherlands)",
"English (Australia)",
"English (Canada)",
"English (Hong Kong SAR)",
"English (India)",
"English (Ireland)",
"English (Kenya)",
"English (New Zealand)",
"English (Nigeria)",
"English (Philippines)",
"English (Singapore)",
"English (South Africa)",
"English (Tanzania)",
"English (UK)",
"English (United States)",
"Estonian (Estonia)",
"Filipino (Philippines)",
"Finnish (Finland)",
"French (Belgium)",
"French (Canada)",
"French (France)",
"French (Switzerland)",
"Galician",
"Georgian (Georgia)",
"German (Austria)",
"German (Germany)",
"German (Switzerland)",
"Greek (Greece)",
"Gujarati (India)",
"Hebrew (Israel)",
"Hindi (India)",
"Hungarian (Hungary)",
"Icelandic (Iceland)",
"Indonesian (Indonesia)",
"Irish (Ireland)",
"Italian (Italy)",
"Japanese (Japan)",
"Javanese (Latin, Indonesia)",
"Kannada (India)",
"Kazakh (Kazakhstan)",
"Khmer (Cambodia)",
"Korean (Korea)",
"Lao (Laos)",
"Latvian (Latvia)",
"Lithuanian (Lithuania)",
"Macedonian (North Macedonia)",
"Malay (Malaysia)",
"Malayalam (India)",
"Maltese (Malta)",
"Marathi (India)",
"Mongolian (Mongolia)",
"Nepali (Nepal)",
"Norwegian Bokm\u00e5l (Norway)",
"Pashto (Afghanistan)",
"Persian (Iran)",
"Polish (Poland)",
"Portuguese (Brazil)",
"Portuguese (Portugal)",
"Romanian (Romania)",
"Russian (Russia)",
"Serbian (Latin, Serbia)",
"Sinhala (Sri Lanka)",
"Slovak (Slovakia)",
"Slovenian (Slovenia)",
"Somali (Somalia)",
"Spanish (Argentina)",
"Spanish (Bolivia)",
"Spanish (Chile)",
"Spanish (Colombia)",
"Spanish (Costa Rica)",
"Spanish (Cuba)",
"Spanish (Dominican Republic)",
"Spanish (Ecuador)",
"Spanish (El Salvador)",
"Spanish (Equatorial Guinea)",
"Spanish (Guatemala)",
"Spanish (Honduras)",
"Spanish (Mexico)",
"Spanish (Nicaragua)",
"Spanish (Panama)",
"Spanish (Paraguay)",
"Spanish (Peru)",
"Spanish (Puerto Rico)",
"Spanish (Spain)",
"Spanish (United States)",
"Spanish (Uruguay)",
"Spanish (Venezuela)",
"Sundanese (Indonesia)",
"Swahili (Kenya)",
"Swahili (Tanzania)",
"Swedish (Sweden)",
"Tamil (India)",
"Tamil (Malaysia)",
"Tamil (Singapore)",
"Tamil (Sri Lanka)",
"Telugu (India)",
"Thai (Thailand)",
"Turkish (T\u00fcrkiye)",
"Ukrainian (Ukraine)",
"Urdu (India)",
"Urdu (Pakistan)",
"Uzbek (Latin, Uzbekistan)",
"Vietnamese (Vietnam)",
"Welsh (United Kingdom)",
"Zulu (South Africa)",
"English - Your Accent",
"English - American Accent"
],
"title": "Language",
"name": "language",
"type": "string",
"description": "The target language in which the video will be translated.",
"default": "Hindi"
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": "heygen-video-translate"
}
}
}
},
{
"name": "wan2.2-animate",
"category": "Video to Video",
"variant": "Anime Video",
"family": "wan2.2",
"group_of": "video",
"description": "Wan2.2 Animate is a video-to-video model for animating a character or replacing a character in existing video clips. It replicates holistic movement and facial expressions from a reference video or pose while preserving the target character\u2019s appearance. You upload both an image (for the character) and a video containing motion/expression, and the model generates a video where the character in your image moves like the reference. Supports 480p or 720p, up to 120 seconds",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Optional prompt for generating video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"mode": {
"enum": [
"animate",
"replace"
],
"title": "Mode",
"name": "mode",
"type": "string",
"description": "Animate Mode: animate the character in input image with movements from the input video. Replace Mode: replace the character in input video with the character in input image.",
"default": "animate"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"image_url",
"video_url"
],
"endpoint_url": "wan2.2-animate"
}
}
}
},
{
"name": "qwen-image-edit-plus",
"category": "Image to Image",
"variant": "Edit Image Plus",
"family": "qwen",
"group_of": "image",
"description": "Qwen Image Edit Plus is an upgraded image-editing model that supports multiple image references and superior text editing. Powered by the 20B-parameter Qwen architecture, it allows changes like background swap, style transfer, object removal/addition, and precise text edits (bilingual: English/Chinese) while maintaining visual consistency and preserving details of the original images.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 3
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "qwen-image-edit-plus"
}
}
}
},
{
"name": "kling-v2.5-turbo-pro-i2v",
"category": "Image to Video",
"variant": "Pro Image to Video",
"family": "kling-v2.5",
"group_of": "video",
"description": "Kling 2.5 Turbo Pro: Top-tier image-to-video generation with unparalleled motion fluidity, cinematic visuals, and exceptional prompt precision.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v2.5-turbo-pro-i2v"
}
}
}
},
{
"name": "kling-v2.5-turbo-pro-t2v",
"category": "Text to Video",
"variant": "Pro Text to Video",
"family": "kling-v2.5",
"group_of": "video",
"description": "Kling 2.5 Turbo Pro: Top-tier text-to-video generation with unparalleled motion fluidity, cinematic visuals, and exceptional prompt precision.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "9:16"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-v2.5-turbo-pro-t2v"
}
}
}
},
{
"name": "wan2.5-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "wan2.5",
"group_of": "video",
"description": "WAN 2.5 Image-to-Video takes your image as the starting frame and turns it into a dynamic video, preserving realism, motion, and camera effects. Upload a static image, add a descriptive text prompt, and the model generates cinematic motion\u2014camera pans, environmental movement, and realistic physics\u2014across the result.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "Audio URL to guide generation (optional).",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "wan2.5-image-to-video"
}
}
}
},
{
"name": "wan2.5-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "wan2.5",
"group_of": "video",
"description": "WAN 2.5 Text-to-Video transforms written prompts into cinematic video clips with dynamic motion, realistic physics, and natural animation. It can also generate characters delivering dialogue, making it ideal for storytelling, ads, and creative showcases.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"audio_url": {
"description": "Audio URL to guide generation (optional).",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.5-text-to-video"
}
}
}
},
{
"name": "wan2.5-image-to-video-fast",
"category": "Image to Video",
"variant": "Image to Video (Fast)",
"family": "wan2.5",
"group_of": "video",
"description": "Convert a single static image into a cinematic short video with realistic motion, dynamic camera movement, and environmental effects. The Fast mode generates high-quality videos quickly, perfect for rapid prototyping, social media clips, and immersive visual storytelling from still images.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "Audio URL to guide generation (optional).",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "wan2.5-image-to-video-fast"
}
}
}
},
{
"name": "wan2.5-text-to-video-fast",
"category": "Text to Video",
"variant": "Text to Video (Fast)",
"family": "wan2.5",
"group_of": "video",
"description": "Transform text prompts into short, cinematic videos with natural motion, realistic environments, and dynamic camera perspectives. Fast mode delivers quick, high-fidelity video generation, ideal for creative storytelling, concept visuals, and social media content.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"audio_url": {
"description": "Audio URL to guide generation (optional).",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.5-text-to-video-fast"
}
}
}
},
{
"name": "wan2.5-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "wan2.5",
"group_of": "image",
"description": "WAN 2.5 Text-to-Image generates high-quality, realistic or stylized images from textual descriptions. It supports detailed visual storytelling, cinematic compositions, and versatile styles \u2014 from portraits and product shots to landscapes and fantasy scenes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 768,
"maxValue": 1440,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1322,
"minValue": 768,
"maxValue": 1440,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.5-text-to-image"
}
}
}
},
{
"name": "topaz-video-upscale",
"category": "Video to Video",
"variant": "Video Upscale",
"family": "topaz",
"group_of": "ai-tools",
"description": "The AI Video Upscaler is a powerful tool designed to enhance the resolution and quality of videos. Whether you're working with low-resolution videos that need a boost or aiming to improve the clarity of existing footage, this upscaler leverages advanced machine learning models to deliver high-quality, upscaled videos.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "The URL of the video to generate the audio for.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"upscale_factor": {
"enum": [
1,
2,
4
],
"title": "Upscale Factor",
"name": "upscale_factor",
"type": "int",
"description": "Factor to upscale the video by (e.g. 2.0 doubles width and height).",
"default": 2
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": "topaz-video-upscale"
}
}
}
},
{
"name": "wan2.5-image-edit",
"category": "Image to Image",
"variant": "Edit Image",
"family": "wan2.5",
"group_of": "image",
"description": "The Wan2.5 Edit Image model allows you to transform existing images with precision and creativity. By providing an image along with an edit prompt, you can make realistic changes, enhancements, or stylistic adjustments\u2014whether it\u2019s altering objects, changing backgrounds, adding details, or applying an entirely new artistic style.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 2
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 2048,
"minValue": 384,
"maxValue": 5000,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 2048,
"minValue": 384,
"maxValue": 5000,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "wan2.5-image-edit"
}
}
}
},
{
"name": "hunyuan-image-3.0",
"category": "Text to Image",
"variant": "Text to Image v3.0",
"family": "hunyuan",
"group_of": "image",
"description": "Hunyuan Image 3.0 brings together powerful architecture (Mixture-of-Experts + autoregressive style) to produce richly detailed and coherent images from complex prompts. It can read narrative descriptions, render text and signage cleanly, and support multiple visual styles \u2014 from photorealism to illustrations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "hunyuan-image-3.0"
}
}
}
},
{
"name": "openai-sora",
"category": "Text to Video",
"variant": "Sora Text to Video",
"family": "sora",
"group_of": "video",
"description": "Sora is a text-to-video generative AI model developed by OpenAI. It can generate short video clips based on descriptive text inputs, producing content that ranges from photorealistic scenes to stylized animations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "openai-sora"
}
}
}
},
{
"name": "openai-sora-2-image-to-video",
"category": "Image to Video",
"variant": "Sora 2 Image to Video",
"family": "sora",
"group_of": "video",
"description": "Sora 2\u2019s I2V lets you bring still images to life by animating them into short video clips with natural motion, audio, and visual effects. While realistic portraits of people aren\u2019t allowed at launch, you can use objects, landscapes, stylized characters or scenes. Use detailed prompts for camera movement, atmosphere, and pacing to get the best results.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 1
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
10,
15
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 10
},
"remove_watermark": {
"type": "boolean",
"title": "Remove Watermark",
"name": "remove_watermark",
"description": "When enabled, removes watermarks from the generated video.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "openai-sora-2-image-to-video"
}
}
}
},
{
"name": "openai-sora-2-text-to-video",
"category": "Text to Video",
"variant": "Sora 2 Text to Video",
"family": "sora",
"group_of": "video",
"description": "Sora 2 T2V converts text prompts into short, dynamic 10-second video clips with synchronized audio. Users can describe scenes, motion, camera angles, and sound effects, and Sora 2 brings them to life with cinematic realism or stylized visuals. Perfect for storytelling, social media content, and creative experimentation, while maintaining high-quality visuals and immersive audio.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
10,
15
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 10
},
"remove_watermark": {
"type": "boolean",
"title": "Remove Watermark",
"name": "remove_watermark",
"description": "When enabled, removes watermarks from the generated video.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "openai-sora-2-text-to-video"
}
}
}
},
{
"name": "ai-video-upscaler-pro",
"category": "Video to Video",
"variant": "Video Upscaler Pro",
"family": "tools",
"group_of": "ai-tools",
"description": "The AI Video Upscaler is a powerful tool designed to enhance the resolution and quality of videos. Whether you're working with low-resolution videos that need a boost or aiming to improve the clarity of existing footage, this upscaler leverages advanced machine learning models to deliver high-quality, upscaled videos.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"resolution": {
"enum": [
"720p",
"1080p",
"2k",
"4k"
],
"description": "Target resolution to upscale.",
"type": "string",
"title": "Resolution",
"name": "resolution",
"default": "720p"
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": "ai-video-upscaler-pro"
}
}
}
},
{
"name": "video-watermark-remover",
"category": "Video to Video",
"variant": "Watermark Remover",
"family": "tools",
"group_of": "ai-tools",
"description": "The AI Video Watermark Remover is our flagship model designed to remove Sora 2 watermarks, logos, captions, and unwanted text from videos without compromising quality. Supporting a wide range of formats, it's fast, efficient, and processes with the highest quality.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": "video-watermark-remover"
}
}
}
},
{
"name": "ovi-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "ovi",
"group_of": "video",
"description": "Ovi is a unified audio\u2013video generation model that can transform a static image plus a descriptive prompt into a short video with synchronized audio. It supports both text-to-video and image-conditioned video inputs. With built-in lip sync, background audio / sound effects, and dialogue support, Ovi brings still visuals to life in cinematic fashion. Videos are generated in 540p resolution.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "ovi-image-to-video"
}
}
}
},
{
"name": "ovi-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "ovi",
"group_of": "video",
"description": "Ovi is a unified model that generates synchronized video and audio from textual input. You write a scene description, including dialogue and ambient sounds, and Ovi produces a short video clip (typically ~5 seconds) where visuals and sound align naturally. Videos are generated in 540p resolution.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "ovi-text-to-video"
}
}
}
},
{
"name": "gpt-5-nano",
"category": "Text to Text",
"variant": "GPT5 Nano Text",
"family": "gpt",
"group_of": "text",
"description": "GPT-5 Nano is a lightweight, high-speed language model from the GPT-5 family designed for instant text generation. It delivers intelligent, context-aware responses for creative writing, summarization, dialogue, code generation, and automation \u2014 all at low latency and cost. Perfect for chatbots, assistants, content tools, and real-time applications that need fast, reliable text output.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the text.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "Text",
"required": [
"prompt"
],
"endpoint_url": "gpt-5-nano"
}
}
}
},
{
"name": "openai-sora-2-pro-image-to-video",
"category": "Image to Video",
"variant": "Sora 2 Pro Image to Video",
"family": "sora",
"group_of": "video",
"description": "Sora 2 Pro I2V brings still images to life, transforming them into short videos with natural motion, realistic lighting, and synchronized audio. Upload your image, describe the movement (camera motion, subject action, ambience), add optional dialogue or sound effects, and watch it animate. Ideal for cinematic reveals, promo videos, social content, or storytelling from a static photo.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 1
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
10,
15,
25
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds. Currently 25 seconds supports 720p only.",
"default": 10
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"remove_watermark": {
"type": "boolean",
"title": "Remove Watermark",
"name": "remove_watermark",
"description": "When enabled, removes watermarks from the generated video.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "openai-sora-2-pro-image-to-video"
}
}
}
},
{
"name": "openai-sora-2-pro-text-to-video",
"category": "Text to Video",
"variant": "Sora 2 Pro Text to Video",
"family": "sora",
"group_of": "video",
"description": "Sora 2 Pro T2V is the high-fidelity version of OpenAI\u2019s video generation model. It converts your text prompts into cinematic, richly detailed video clips with synchronized audio, realistic motion, strong physics, and creative control over style, mood, and pacing. Perfect for creators, storytellers, advertisers, and anyone who wants top-quality video content from text.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
10,
15,
25
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds. Currently 25 seconds supports 720p only.",
"default": 10
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"remove_watermark": {
"type": "boolean",
"title": "Remove Watermark",
"name": "remove_watermark",
"description": "When enabled, removes watermarks from the generated video.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "openai-sora-2-pro-text-to-video"
}
}
}
},
{
"name": "higgsfield-soul-image-to-image",
"category": "Image to Image",
"variant": "Image to Image",
"family": "higgsfield",
"group_of": "image",
"description": "SOUL is an AI image model focused on hyper-realistic, magazine or editorial-style visuals, especially for fashion, portraits, lifestyle, and commercial content. It offers over 50 curated style presets to get a specific aesthetic without needing complicated prompt engineering. It generates photography-quality images with lighting, textures, and context that feel real \u2014 including natural imperfections like film grain, dust, or lens effects for authenticity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image (max 1500 characters).",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"style": {
"enum": [
"Creatures",
"Medieval",
"Spotlight",
"Giant People",
"Red balloon",
"green editorial",
"Subway",
"Library",
"Realistic",
"DigitalCam",
"Grillz Selfie",
"Bleached Brows",
"Sitting on the Street",
"Crossing the street",
"Angel Wings",
"Duplicate",
"cocktail",
"Quiet luxury",
"Fireproof",
"Elevator Mirror",
"360 cam",
"Glitch",
"FashionShow",
"PixeletedFace",
"Sunbathing",
"Paper Face",
"90s Grain",
"Geominimal",
"Foggy Morning",
"Overexposed",
"Sunset beach",
"Giant Accessory",
"RingSelfie",
"Street view",
"90\u2019s Editorial",
"Rhyme & blues",
"2000s Cam",
"CCTV",
"0.5 Outfit",
"Amalfi Summer",
"Bimbocore",
"0.5 Selfie",
"Sand",
"Vintage PhotoBooth",
"afterparty cam",
"Babydoll MakeUp",
"Through The Glass",
"Gallery",
"Eating Food",
"Swords Hill",
"Office beach",
"Help It's Too Big",
"Japandi",
"iPhone",
"Gorpcore",
"Indie sleaze",
"Fairycore",
"Tumblr",
"Avant-garde",
"HairClips",
"birthday mess",
"Clouded Dream",
"Y2K Posters",
"tokyo drift",
"Object Makeup",
"Graffiti",
"Sunburnt",
"hallway noir",
"2000s Fashion",
"Night Beach",
"Movie",
"Long legs",
"7\\",
"General",
"Nail Check",
"Coquette core",
"Mixed Media",
"Selfcare",
"Grunge",
"Double take",
"505room",
"Flight mode",
"Escalator",
"burgundy suit",
"Fisheye",
"Shoe Check",
"Rainy Day",
"Mt. Fuji",
"Sea breeze",
"Invertethereal",
"Y2K",
"Tokyo Streetstyle",
"chrome exit",
"Night rider",
"Artwork",
"Glazed doll skin makeup",
"mount view",
"2049",
"blackout fit",
"Bike mafia",
"static glow",
"Nicotine glow",
"brick shade",
"dmv",
"Fish-eye twin",
"It\u2019s french"
],
"title": "Style",
"name": "style",
"type": "string",
"description": "Choose preset for soul image generation.",
"default": "DigitalCam"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"4:5",
"5:4",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "9:16"
},
"strength": {
"title": "Strength",
"name": "strength",
"type": "int",
"description": "The strength to use for the style.",
"default": 0.5,
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"quality": {
"enum": [
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The resolution of the output image.",
"default": "medium"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "higgsfield-soul-image-to-image"
},
"examples": [
{
"name": "0.5 Outfit",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/0.5Outfit.jpg"
},
{
"name": "0.5 Selfie",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/0.5Selfie.jpg"
},
{
"name": "2000s Cam",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/2000sCam.jpg"
},
{
"name": "2000s Fashion",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/2000sFashion.webp"
},
{
"name": "2049",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/2049.webp"
},
{
"name": "360 cam",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/360cam.jpg"
},
{
"name": "505room",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/505room.jpg"
},
{
"name": "7\\",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/7.webp"
},
{
"name": "90s Grain",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/90sGrain.jpg"
},
{
"name": "90\u2019s Editorial",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/90\u2019sEditorial.webp"
},
{
"name": "afterparty cam",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/afterpartycam.webp"
},
{
"name": "Amalfi Summer",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/AmalfiSummer.webp"
},
{
"name": "Angel Wings",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/AngelWings.webp"
},
{
"name": "Artwork",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Artwork.webp"
},
{
"name": "Avant-garde",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Avant-garde.jpg"
},
{
"name": "Babydoll MakeUp",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/BabydollMakeUp.jpg"
},
{
"name": "Bike mafia",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Bikemafia.webp"
},
{
"name": "Bimbocore",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Bimbocore.jpg"
},
{
"name": "birthday mess",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/birthdaymess.webp"
},
{
"name": "blackout fit",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/blackoutfit.webp"
},
{
"name": "Bleached Brows",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/BleachedBrows.jpg"
},
{
"name": "brick shade",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/brickshade.webp"
},
{
"name": "burgundy suit",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/burgundysuit.webp"
},
{
"name": "CCTV",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/CCTV.webp"
},
{
"name": "chrome exit",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/chromeexit.webp"
},
{
"name": "Clouded Dream",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/CloudedDream.webp"
},
{
"name": "cocktail",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/cocktail.webp"
},
{
"name": "Coquette core",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Coquettecore.jpg"
},
{
"name": "Creatures",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Creatures.webp"
},
{
"name": "Crossing the street",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Crossingthestreet.webp"
},
{
"name": "DigitalCam",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/DigitalCam.webp"
},
{
"name": "dmv",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/dmv.webp"
},
{
"name": "Double take",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Doubletake.webp"
},
{
"name": "Duplicate",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Duplicate.jpg"
},
{
"name": "Eating Food",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/EatingFood.jpg"
},
{
"name": "Elevator Mirror",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/ElevatorMirror.jpg"
},
{
"name": "Escalator",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Escalator.webp"
},
{
"name": "Fairycore",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Fairycore.webp"
},
{
"name": "FashionShow",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/FashionShow.jpg"
},
{
"name": "Fireproof",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Fireproof.webp"
},
{
"name": "Fish Eyetwin",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Fish-eyetwin.webp"
},
{
"name": "Fisheye",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Fisheye.webp"
},
{
"name": "Flight mode",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Flightmode.webp"
},
{
"name": "Foggy Morning",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/FoggyMorning.jpg"
},
{
"name": "Gallery",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Gallery.webp"
},
{
"name": "General",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/General.jpg"
},
{
"name": "Geominimal",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Geominimal.webp"
},
{
"name": "Giant Accessory",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/GiantAccessory.jpg"
},
{
"name": "Giant People",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/GiantPeople.webp"
},
{
"name": "Glazed doll skin makeup",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Glazeddollskinmakeup.jpg"
},
{
"name": "Glitch",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Glitch.webp"
},
{
"name": "Gorpcore",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Gorpcore.jpg"
},
{
"name": "Graffiti",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Graffiti.webp"
},
{
"name": "green editorial",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/greeneditorial.webp"
},
{
"name": "Grillz Selfie",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/GrillzSelfie.webp"
},
{
"name": "Grunge",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Grunge.jpg"
},
{
"name": "HairClips",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/HairClips.jpg"
},
{
"name": "hallway noir",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/hallwaynoir.webp"
},
{
"name": "Help It's Too Big",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/HelpIt'sTooBig.webp"
},
{
"name": "Indie sleaze",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Indiesleaze.jpg"
},
{
"name": "Invertethereal",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Invertethereal.jpg"
},
{
"name": "iPhone",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/iPhone.jpg"
},
{
"name": "It\u2019s french",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/It\u2019sfrench.webp"
},
{
"name": "Japandi",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Japandi.jpg"
},
{
"name": "Library",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Library.webp"
},
{
"name": "Long legs",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Longlegs.webp"
},
{
"name": "Medieval",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Medieval.jpg"
},
{
"name": "Mixed Media",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/MixedMedia.webp"
},
{
"name": "mount view",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/mountview.webp"
},
{
"name": "Movie",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Movie.webp"
},
{
"name": "Mt. Fuji",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Mt.Fuji.webp"
},
{
"name": "Nail Check",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/NailCheck.jpg"
},
{
"name": "Nicotineglow",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Nicotineglow.webp"
},
{
"name": "Night Beach",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/NightBeach.jpg"
},
{
"name": "Night rider",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Nightrider.webp"
},
{
"name": "Object Makeup",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/ObjectMakeup.webp"
},
{
"name": "Office beach",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Officebeach.jpg"
},
{
"name": "Overexposed",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Overexposed.jpg"
},
{
"name": "Paper Face",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/PaperFace.webp"
},
{
"name": "PixeletedFace",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/PixeletedFace.webp"
},
{
"name": "Quiet luxury",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Quietluxury.webp"
},
{
"name": "Rainy Day",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/RainyDay.jpg"
},
{
"name": "Realistic",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Realistic.jpg"
},
{
"name": "Red balloon",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Redballoon.webp"
},
{
"name": "Rhyme & blues",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Rhyme&blues.webp"
},
{
"name": "RingSelfie",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/RingSelfie.jpg"
},
{
"name": "Sand",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Sand.jpg"
},
{
"name": "Sea breeze",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Seabreeze.webp"
},
{
"name": "Selfcare",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Selfcare.jpg"
},
{
"name": "Shoe Check",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/ShoeCheck.jpg"
},
{
"name": "Sitting on the Street",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/SittingontheStreet.jpg"
},
{
"name": "Spotlight",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Spotlight.webp"
},
{
"name": "Static glow",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/staticglow.webp"
},
{
"name": "Street view",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Streetview.webp"
},
{
"name": "Subway",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Subway.webp"
},
{
"name": "Sunbathing",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Sunbathing.webp"
},
{
"name": "Sunburnt",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Sunburnt.webp"
},
{
"name": "Sunset beach",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Sunsetbeach.webp"
},
{
"name": "Swords Hill",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/SwordsHill.jpg"
},
{
"name": "Through The Glass",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/ThroughTheGlass.jpg"
},
{
"name": "tokyo drift",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/tokyodrift.webp"
},
{
"name": "Tokyo Streetstyle",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/TokyoStreetstyle.jpg"
},
{
"name": "Tumblr",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Tumblr.jpg"
},
{
"name": "Vintage PhotoBooth",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/VintagePhotoBooth.jpg"
},
{
"name": "Y2K",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Y2K.jpg"
},
{
"name": "Y2K Posters",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Y2KPosters.webp"
}
]
}
}
},
{
"name": "leonardoai-phoenix-1.0",
"category": "Text to Image",
"variant": "Phoenix 1.0 T2I",
"family": "leonardoai",
"group_of": "image",
"description": "LeonardoAI Phoenix 1.0 is a professional-grade AI image model designed for realistic, cinematic, and highly detailed visuals. It excels at interpreting complex prompts, rendering text within images, and creating high-resolution outputs suitable for editorial, commercial, or creative projects.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"4:5",
"5:4",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "leonardoai-phoenix-1.0"
}
}
}
},
{
"name": "leonardoai-lucid-origin",
"category": "Text to Image",
"variant": "Lucid Origin T2I",
"family": "leonardoai",
"group_of": "image",
"description": "Lucid Origin is LeonardoAI\u2019s advanced image generation model, designed for ultra-realistic, vibrant, and highly detailed visuals. It excels at creating photorealistic portraits, landscapes, product shots, and stylized art while faithfully following complex prompts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"3:4",
"4:3",
"4:5",
"5:4",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "leonardoai-lucid-origin"
}
}
}
},
{
"name": "leonardoai-motion-2.0",
"category": "Image to Video",
"variant": "Motion 2.0 I2V",
"family": "leonardoai",
"group_of": "video",
"description": "Motion 2.0 is Leonardo.AI's cutting-edge model for creating high-quality 5-second videos from text prompts. It offers enhanced control over animation, including camera movements, lighting, and scene dynamics.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image (optional).",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "leonardoai-motion-2.0"
}
}
}
},
{
"name": "higgsfield-dop-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "higgsfield",
"group_of": "video",
"description": "Higgsfield\u2019s DOP (Director of Photography) Motion Effects empower creators to combine cinematic camera moves with built-in visual effects\u2014like explosions, fire, distortion, disintegration, and transitions\u2014directly in AI video generation. You choose from a library of motion presets (e.g. Earth Zoom, Bullet Time, Dolly Zoom) and overlay dynamic effects that accentuate storytelling without needing a full VFX pipeline.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"motion": {
"enum": [
"360 Orbit",
"3D Rotation",
"Abstract",
"Action Run",
"Agent Reveal",
"Angel Wings",
"Arc Left",
"Arc Right",
"Baseball Kick",
"Basketball Dunks",
"Black Tears",
"Bloom Mouth",
"Boxing",
"Buckle Up",
"Building Explosion",
"Bullet Time",
"Car Chasing",
"Car Explosion",
"Car Grip",
"Catwalk",
"Clone Explosion",
"Crane Down",
"Crane Over The Head",
"Crane Up",
"Crash Zoom In",
"Crash Zoom Out",
"Datamosh",
"Diamond",
"Dirty Lens",
"Disintegration",
"Dolly In",
"Dolly Left",
"Dolly Out",
"Dolly Right",
"Dolly Zoom In",
"Dolly Zoom Out",
"Double Dolly",
"Downhill POV",
"Duplicate",
"Dutch Angle",
"Earth Zoom Out",
"Eyes In",
"Face Punch",
"Fire Breathe",
"Fisheye",
"Floating Fish",
"Flood",
"Floral Eyes",
"Flying",
"Focus Change",
"FPV Drone",
"Freezing",
"Garden Bloom",
"General",
"Glam",
"Glowing Fish",
"Glowshift",
"Handheld",
"Head Explosion",
"Head Off",
"Head Tracking",
"Hyperlapse",
"Incline",
"Innerlight",
"Invisible",
"Jelly Drift",
"Jib Down",
"Jib Up",
"Kiss",
"Lazy Susan",
"Lens Crack",
"Lens Flare",
"Levitation",
"Low Shutter",
"Medusa Gorgona",
"Melting",
"Moonwalk Left",
"Moonwalk Right",
"Morphskin",
"Mouth In",
"Object POV",
"Overhead",
"Paint Splash",
"Paparazzi",
"Powder Explosion",
"Push To Glass",
"Rap Flex",
"Robo Arm",
"Roll Transition",
"Sand Storm",
"Set on Fire",
"Skateboard Glide",
"Skateboard Ollie",
"Skate Cruise",
"Ski Carving",
"Skin Surge",
"Ski Powder",
"Snorricam",
"Snowboard Carving",
"Snowboard Powder",
"Soul Jump",
"Static",
"Super 8MM",
"Super Dolly In",
"Super Dolly Out",
"Tentacles",
"Through Object In",
"Through Object Out",
"Thunder God",
"Tilt Down",
"Tilt up",
"Timelapse Human",
"Timelapse Landscape",
"Turning Metal",
"VHS",
"Whip Pan",
"Wiggle",
"Wind to Face",
"YoYo Zoom",
"Zoom In",
"Zoom Out"
],
"title": "Motion",
"name": "motion",
"type": "string",
"description": "Terminoogies to use for transform.",
"default": "Bullet Time"
},
"strength": {
"title": "Strength",
"name": "strength",
"type": "int",
"description": "The strength to use for the motion.",
"default": 1,
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"options": {
"enum": [
"dop-lite",
"dop-turbo",
"dop-preview"
],
"title": "Options",
"name": "options",
"type": "string",
"description": "Model versions.",
"default": "dop-lite"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url",
"motion"
],
"endpoint_url": "higgsfield-dop-image-to-video"
},
"examples": [
{
"name": "360 Orbit",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/360+Orbit.mp4"
},
{
"name": "3D Rotation",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/3D+Rotation.mp4"
},
{
"name": "Action Run",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Action+Run.mp4"
},
{
"name": "Agent Reveal",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Agent+Reveal.mp4"
},
{
"name": "Angel Wings",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Angel+Wings.mp4"
},
{
"name": "Arc Left",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Arc+Left.mp4"
},
{
"name": "Arc Right",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Arc+Right.mp4"
},
{
"name": "Baseball Kick",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Baseball+Kick.mp4"
},
{
"name": "Basketball Dunks",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Basketball+Dunks.mp4"
},
{
"name": "Black Tears",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Black+Tears.mp4"
},
{
"name": "Bloom Mouth",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Bloom+Mouth.mp4"
},
{
"name": "Boxing",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Boxing.mp4"
},
{
"name": "Buckle Up",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Buckle+Up.mp4"
},
{
"name": "Building Explosion",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Building+Explosion.mp4"
},
{
"name": "Bullet Time",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Bullet+Time.mp4"
},
{
"name": "Car Chasing",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Car+Chasing.mp4"
},
{
"name": "Car Explosion",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Car+Explosion.mp4"
},
{
"name": "Car Grip",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Car+Grip.mp4"
},
{
"name": "Catwalk",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Catwalk.mp4"
},
{
"name": "Clone Explosion",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Clone+Explosion.mp4"
},
{
"name": "Crane Down",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Crane+Down.mp4"
},
{
"name": "Crane Over The Head",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Crane+Over+The+Head.mp4"
},
{
"name": "Crane Up",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Crane+Up.mp4"
},
{
"name": "Crash Zoom In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Crash+Zoom+In.mp4"
},
{
"name": "Crash Zoom Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Crash+Zoom+Out.mp4"
},
{
"name": "Datamosh",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Datamosh.mp4"
},
{
"name": "Diamond",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Diamond.mp4"
},
{
"name": "Dirty Lens",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dirty+Lens.mp4"
},
{
"name": "Disintegration",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Disintegration.mp4"
},
{
"name": "Dolly In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dolly+In.mp4"
},
{
"name": "Dolly Left",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dolly+Left.mp4"
},
{
"name": "Dolly Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dolly+Out.mp4"
},
{
"name": "Dolly Right",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dolly+Right.mp4"
},
{
"name": "Dolly Zoom In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dolly+Zoom+In.mp4"
},
{
"name": "Dolly Zoom Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dolly+Zoom+Out.mp4"
},
{
"name": "Double Dolly",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Double+Dolly.mp4"
},
{
"name": "Downhill POV",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Downhill+POV.mp4"
},
{
"name": "Duplicate",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Duplicate.mp4"
},
{
"name": "Dutch Angle",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Dutch+Angle.mp4"
},
{
"name": "Earth Zoom Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Earth+Zoom+Out.mp4"
},
{
"name": "Eyes In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Eyes+In.mp4"
},
{
"name": "Face Punch",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Face+Punch.mp4"
},
{
"name": "Fire Breathe",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Fire+Breathe.mp4"
},
{
"name": "Fisheye",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Fisheye.mp4"
},
{
"name": "Floating Fish",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Floating+Fish.mp4"
},
{
"name": "Flood",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Flood.mp4"
},
{
"name": "Floral Eyes",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Floral+Eyes.mp4"
},
{
"name": "Flying",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Flying.mp4"
},
{
"name": "Focus Change",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Focus+Change.mp4"
},
{
"name": "FPV Drone",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/FPV+Drone.mp4"
},
{
"name": "Freezing",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Freezing.mp4"
},
{
"name": "Garden Bloom",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Garden+Bloom.mp4"
},
{
"name": "General",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/General.mp4"
},
{
"name": "Glam",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Glam.mp4"
},
{
"name": "Glowing Fish",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Glowing+Fish.mp4"
},
{
"name": "Glowshift",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Glowshift.mp4"
},
{
"name": "Handheld",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Handheld.mp4"
},
{
"name": "Head Explosion",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Head+Explosion.mp4"
},
{
"name": "Head Off",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Head+Off.mp4"
},
{
"name": "Head Tracking",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Head+Tracking.mp4"
},
{
"name": "Hyperlapse",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Hyperlapse.mp4"
},
{
"name": "Incline",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Incline.mp4"
},
{
"name": "Innerlight",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Innerlight.mp4"
},
{
"name": "Invisible",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Invisible.mp4"
},
{
"name": "Jelly Drift",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Jelly+Drift.mp4"
},
{
"name": "Jib Down",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Jib+Down.mp4"
},
{
"name": "Jib Up",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Jib+Up.mp4"
},
{
"name": "Kiss",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Kiss.mp4"
},
{
"name": "Lazy Susan",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Lazy+Susan.mp4"
},
{
"name": "Lens Crack",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Lens+Crack.mp4"
},
{
"name": "Lens Flare",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Lens+Flare.mp4"
},
{
"name": "Levitation",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Levitation.mp4"
},
{
"name": "Low Shutter",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Low+Shutter.mp4"
},
{
"name": "Medusa Gorgona",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Medusa+Gorgona.mp4"
},
{
"name": "Melting",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Melting.mp4"
},
{
"name": "Moonwalk Left",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Moonwalk+Left.mp4"
},
{
"name": "Moonwalk Right",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Moonwalk+Right.mp4"
},
{
"name": "Morphskin",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Morphskin.mp4"
},
{
"name": "Mouth In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Mouth+In.mp4"
},
{
"name": "Object POV",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Object+POV.mp4"
},
{
"name": "Overhead",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Overhead.mp4"
},
{
"name": "Paint Splash",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Paint+Splash.mp4"
},
{
"name": "Paparazzi",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Paparazzi.mp4"
},
{
"name": "Powder Explosion",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Powder+Explosion.mp4"
},
{
"name": "Push To Glass",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Push+To+Glass.mp4"
},
{
"name": "Rap Flex",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Rap+Flex.mp4"
},
{
"name": "Robo Arm",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Robo+Arm.mp4"
},
{
"name": "Roll Transition",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Roll+Transition.mp4"
},
{
"name": "Sand Storm",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Sand+Storm.mp4"
},
{
"name": "Set on Fire",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Set+on+Fire.mp4"
},
{
"name": "Skate Cruise",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Skate+Cruise.mp4"
},
{
"name": "Skateboard Glide",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Skateboard+Glide.mp4"
},
{
"name": "Skateboard Ollie",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Skateboard+Ollie.mp4"
},
{
"name": "Ski Carving",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Ski+Carving.mp4"
},
{
"name": "Ski Powder",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Ski+Powder.mp4"
},
{
"name": "Skin Surge",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Skin+Surge.mp4"
},
{
"name": "Snorricam",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Snorricam.mp4"
},
{
"name": "Snowboard Carving",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Snowboard+Carving.mp4"
},
{
"name": "Snowboard Powder",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Snowboard+Powder.mp4"
},
{
"name": "Soul Jump",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Soul+Jump.mp4"
},
{
"name": "Static",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Static.mp4"
},
{
"name": "Super Dolly In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Super+Dolly+In.mp4"
},
{
"name": "Super Dolly Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Super+Dolly+Out.mp4"
},
{
"name": "Tentacles",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Tentacles.mp4"
},
{
"name": "Through Object In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Through+Object+In.mp4"
},
{
"name": "Through Object Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Through+Object+Out.mp4"
},
{
"name": "Thunder God",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Thunder+God.mp4"
},
{
"name": "Tilt Down",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Tilt+Down.mp4"
},
{
"name": "Tilt up",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Tilt+up.mp4"
},
{
"name": "Timelapse Human",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Timelapse+Human.mp4"
},
{
"name": "Timelapse Landscape",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Timelapse+Landscape.mp4"
},
{
"name": "Turning Metal",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Turning+Metal.mp4"
},
{
"name": "Whip Pan",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Whip+Pan.mp4"
},
{
"name": "Wiggle",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Wiggle.mp4"
},
{
"name": "Wind to Face",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Wind+to+Face.mp4"
},
{
"name": "YoYo Zoom",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/YoYo+Zoom.mp4"
},
{
"name": "Zoom In",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Zoom+In.mp4"
},
{
"name": "Zoom Out",
"url": "https://d3adwkbyhxyrtq.cloudfront.net/webassets/videomodels/Zoom+Out.mp4"
}
]
}
}
},
{
"name": "remix-video",
"category": "Video to Video",
"variant": "Remix Video",
"family": "tools",
"group_of": "video",
"description": "Transform and resize your videos effortlessly with remix video tool.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "URL of the input video to remix. Must be less than 20mb and 10 seconds.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "9:16"
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": "remix-video"
}
}
}
},
{
"name": "veo3.1-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "veo3.1",
"group_of": "video",
"description": "Veo 3.1 is Google's advanced AI video generation model that allows users to create high-quality, 8-second videos from static images. This feature is particularly useful for transforming concept art, storyboards, or static visuals into dynamic video clips with synchronized audio.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 8
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "veo3.1-image-to-video"
}
}
}
},
{
"name": "veo3.1-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "veo3.1",
"group_of": "video",
"description": "Veo 3.1 is Google's advanced AI video generation model that transforms text prompts into high-quality videos. This model offers enhanced realism, richer audio, and improved narrative control, making it suitable for creators seeking cinematic-quality content.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 8
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "veo3.1-text-to-video"
}
}
}
},
{
"name": "veo3.1-fast-image-to-video",
"category": "Image to Video",
"variant": "Image to Video [Fast]",
"family": "veo3.1",
"group_of": "video",
"description": "Veo 3.1 Fast is an optimized version of Google\u2019s Veo 3.1 AI that transforms static images into dynamic 8-second videos at higher speed. It preserves visual fidelity while enabling rapid generation, making it ideal for social media clips, storyboards, and quick creative previews.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 8
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "veo3.1-fast-image-to-video"
}
}
}
},
{
"name": "veo3.1-fast-text-to-video",
"category": "Text to Video",
"variant": "Text to Video [Fast]",
"family": "veo3.1",
"group_of": "video",
"description": "Veo 3.1 Fast T2V is a high-speed AI video model that transforms text prompts into realistic 8-second videos. It emphasizes rapid generation while maintaining visual quality, accurate scene representation, and smooth motion. Ideal for social media, creative storytelling, or rapid concept visualization, it supports cinematic framing, dynamic lighting, and natural object movements.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 8
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "veo3.1-fast-text-to-video"
}
}
}
},
{
"name": "veo3.1-reference-to-video",
"category": "Image to Video",
"variant": "Reference to Video",
"family": "veo3.1",
"group_of": "video",
"description": "Veo 3.1 R2V allows creators to generate dynamic videos using up to three reference images. The model maintains visual consistency of characters, objects, and style throughout the video, producing cinematic-quality 8-second clips. It\u2019s perfect for turning concept art, storyboards, or character designs into short, animated sequences while preserving original aesthetics.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 3
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"enum": [
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 8
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "veo3.1-reference-to-video"
}
}
}
},
{
"name": "sora2-storyboard",
"category": "Text to Text",
"variant": "Storyboard",
"family": "sora2",
"group_of": "ai-tools",
"description": "Sora 2 Storyboard",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"duration": {
"enum": [
10,
15
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated scenes in seconds",
"default": 10
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "create-story-board-data"
}
}
}
},
{
"name": "openai-sora-2-pro-storyboard",
"category": "Text to Video",
"variant": "Sora 2 Pro Storyboard",
"family": "sora",
"group_of": "video",
"description": "Sora 2 Pro enables creators to structure video narratives by chaining multiple scenes through storyboard \u201ccards.\u201d Each card defines a segment of the video\u2014setting, characters, actions, timing\u2014and the model stitches them into a cohesive multi-scene video. This gives you more control over pacing, transitions, and storytelling flow.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"shots": {
"title": "Shots",
"name": "shots",
"type": "array",
"items": {
"type": "object",
"properties": {
"scene": {
"type": "string",
"format": "text",
"title": "Scene",
"name": "scene",
"description": "Scene description/prompt."
},
"duration": {
"type": "number",
"name": "duration",
"title": "Duration",
"description": "Duration in seconds.",
"minValue": 0,
"maxValue": 10,
"step": 0.01,
"default": 1
}
}
},
"description": "Array of scene objects defining the storyboard sequence. Each scene contains a duration and description.",
"maxItems": 30
},
"duration": {
"enum": [
10,
15,
25
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 10
},
"images_list": {
"description": "Upload or provide image urls.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 1
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "9:16"
}
},
"title": "BaseInput",
"required": [
"shots"
],
"endpoint_url": "openai-sora-2-pro-storyboard"
}
}
}
},
{
"name": "veo3.1-extend-video",
"category": "Text to Video",
"variant": "Extend Video",
"family": "veo3.1",
"group_of": "video",
"description": "Veo 3.1\u2019s Extend Video mode lets you continue or expand an existing video clip seamlessly. Starting from a short generated video, you can prompt the model to extend the scene\u2014keeping visual style, characters, motion, and audio consistent. This model needs original task_id of the video.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"request_id": {
"description": "Request ID of the original video generation. Must be a valid Id returned from the video generation interface.",
"format": "text",
"type": "string",
"title": "Request Id",
"name": "request_id",
"placeholder": "abcdefg-123-456-789-a1b2c3d4e5f6"
},
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
}
},
"title": "BaseInput",
"required": [
"request_id",
"prompt"
],
"endpoint_url": "veo3.1-extend-video"
}
}
}
},
{
"name": "gpt-5-mini",
"category": "Text to Text",
"variant": "GPT5 Mini Text",
"family": "gpt",
"group_of": "text",
"description": "GPT\u20115 Mini is a compact yet powerful AI that converts plain text ideas into detailed, structured prompts suitable for use in text-to-image, text-to-video, and other generative AI models. It\u2019s perfect for creators who want to quickly craft high-quality prompts without manually thinking about style, composition, and descriptive details. The model helps accelerate workflows for artists, video producers, and designers.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the text.",
"type": "string",
"title": "Prompt",
"name": "prompt"
}
},
"title": "Text",
"required": [
"prompt"
],
"endpoint_url": "gpt-5-mini"
}
}
}
},
{
"name": "seedance-pro-i2v-fast",
"category": "Image to Video",
"variant": "Pro Image to Video Fast",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Pro Fast is the high-speed image-to-video generation variant from ByteDance\u2019s Seedance series. With this model you upload a reference image and\u2014using a text prompt\u2014generate short, dynamic video clips (typically 3-12 seconds) featuring smooth motion, cinematic camera moves, prompt-accurate actions, and high visual fidelity. It supports resolutions up to 1080p, multiple aspect ratios (16:9, 9:16, etc.), and rapid turnaround\u2014ideal for social content, product motion, storytelling from a still, and fast prototyping.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 2,
"maxValue": 12,
"step": 1
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "seedance-pro-i2v-fast"
}
}
}
},
{
"name": "seedance-pro-t2v-fast",
"category": "Text to Video",
"variant": "Pro Text to Video Fast",
"family": "bytedance",
"group_of": "video",
"description": "Seedance Pro Fast is ByteDance\u2019s advanced text-to-video model that turns natural-language prompts into short, cinematic video clips with realistic motion, camera dynamics, and consistent scene detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 2,
"maxValue": 12,
"step": 1
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-pro-t2v-fast"
}
}
}
},
{
"name": "ltx-2-pro-image-to-video",
"category": "Image to Video",
"variant": "Pro Image to Video",
"family": "ltx",
"group_of": "video",
"description": "LTX-2 Pro is the high-fidelity video-generation engine by Lightricks designed for professional workflows, supporting both text-to-video and image-to-video inputs. It enables realistic motion, synchronized audio-video, cinematic camera moves and stylized visuals. Ideal for your timeline-based video interface: you supply a prompt or image, define duration/aspect ratio, then it generates a clip that you can ingest, rename, batch-move, split or timeline-edit.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"enum": [
6,
8,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "ltx-2-pro-image-to-video"
}
}
}
},
{
"name": "ltx-2-pro-text-to-video",
"category": "Text to Video",
"variant": "Pro Text to Video",
"family": "ltx",
"group_of": "video",
"description": "LTX-2 Pro is the high-fidelity video-generation engine by Lightricks designed for professional workflows, supporting both text-to-video and image-to-video inputs. It enables realistic motion, synchronized audio-video, cinematic camera moves and stylized visuals. Ideal for your timeline-based video interface: you supply a prompt or image, define duration/aspect ratio, then it generates a clip that you can ingest, rename, batch-move, split or timeline-edit.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"duration": {
"enum": [
6,
8,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "ltx-2-pro-text-to-video"
}
}
}
},
{
"name": "ltx-2-fast-image-to-video",
"category": "Image to Video",
"variant": "Fast Image to Video",
"family": "ltx",
"group_of": "video",
"description": "LTX-2 Fast is a speed-optimized mode of the LTX-2 engine by Lightricks, focused on generating short video clips from a still image + prompt (I2V) with good fidelity and rapid turnaround. It supports audio/video together, multiple aspect ratios, and is ideal when you need quick output for iteration or storyboarding.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"enum": [
6,
8,
10,
12,
14,
16,
18,
20
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "ltx-2-fast-image-to-video"
}
}
}
},
{
"name": "ltx-2-fast-text-to-video",
"category": "Text to Video",
"variant": "Fast Text to Video",
"family": "ltx",
"group_of": "video",
"description": "LTX Video Fast is a speed-optimised mode of Lightricks\u2019 video-generation engine, supporting text-to-video workflows. It allows you to input a descriptive prompt and get a short video clip with motion, camera movement, lighting, and stylised visuals. The underlying model (LTX-Video) is built for real-time or near-real-time generation of video clips.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"duration": {
"enum": [
6,
8,
10,
12,
14,
16,
18,
20
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "ltx-2-fast-text-to-video"
}
}
}
},
{
"name": "vidu-q2-reference",
"category": "Image to Video",
"variant": "Reference I2V",
"family": "vidu-q2",
"group_of": "video",
"description": "Vidu Q2 Reference Video generates breathtaking cinematic clips from text prompts guided by multiple reference images. Each image refines the model\u2019s understanding of subject, environment, and visual tone \u2014 ensuring perfect consistency in appearance and motion across every frame.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 7
},
"resolution": {
"enum": [
"360p",
"540p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"4:3",
"3:4",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 2,
"maxValue": 8,
"step": 1
},
"movement_amplitude": {
"enum": [
"auto",
"small",
"medium",
"large"
],
"title": "Movement Amplitude",
"name": "movement_amplitude",
"type": "string",
"description": "The movement amplitude of objects in the frame.",
"default": "auto"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "vidu-q2-reference"
}
}
}
},
{
"name": "vidu-q2-turbo-start-end-video",
"category": "Image to Video",
"variant": "Turbo I2V",
"family": "vidu-q2",
"group_of": "video",
"description": "Vidu Q2 Turbo Start\u2013End Video creates highly detailed cinematic sequences by interpolating between two visual states \u2014 your start frame and end frame. Built for story moments, cinematic transformations, product reveals, and artistic transitions, it captures smooth motion, realistic lighting shifts, and dynamic camera movements while maintaining fidelity and emotional tone.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 2,
"maxValue": 8,
"step": 1
},
"bgm": {
"type": "boolean",
"title": "Bgm",
"name": "bgm",
"description": "The background music for generating the output.",
"default": true
},
"movement_amplitude": {
"enum": [
"auto",
"small",
"medium",
"large"
],
"title": "Movement Amplitude",
"name": "movement_amplitude",
"type": "string",
"description": "The movement amplitude of objects in the frame.",
"default": "auto"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url",
"last_image"
],
"endpoint_url": "vidu-q2-turbo-start-end-video"
}
}
}
},
{
"name": "vidu-q2-pro-start-end-video",
"category": "Image to Video",
"variant": "Pro I2V",
"family": "vidu-q2",
"group_of": "video",
"description": "Vidu Q2 Pro Start\u2013End Video is a professional-grade model built for cinematic transformation storytelling. It evolves a scene, subject, or concept from one moment to another through smooth visual interpolation, natural lighting transitions, and dynamic motion.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 2,
"maxValue": 8,
"step": 1
},
"bgm": {
"type": "boolean",
"title": "Bgm",
"name": "bgm",
"description": "The background music for generating the output.",
"default": true
},
"movement_amplitude": {
"enum": [
"auto",
"small",
"medium",
"large"
],
"title": "Movement Amplitude",
"name": "movement_amplitude",
"type": "string",
"description": "The movement amplitude of objects in the frame.",
"default": "auto"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url",
"last_image"
],
"endpoint_url": "vidu-q2-pro-start-end-video"
}
}
}
},
{
"name": "minimax-hailuo-2.3-pro-i2v",
"category": "Image to Video",
"variant": "Pro I2V",
"family": "minimax-2.3",
"group_of": "video",
"description": "Hailuo 2.3 Pro I2V breathes life into still images with stunning motion synthesis and cinematic camera control. Using deep motion understanding, it predicts realistic subject movement, depth, and environmental motion from a single input frame \u2014 delivering smooth, film-grade clips.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "minimax-hailuo-2.3-pro-i2v"
}
}
}
},
{
"name": "minimax-hailuo-2.3-pro-t2v",
"category": "Text to Video",
"variant": "Pro T2V",
"family": "minimax-2.3",
"group_of": "video",
"description": "Hailuo 2.3 Pro T2V turns your imagination into motion-picture realism. It interprets natural language prompts and generates visually stunning cinematic sequences that capture depth, atmosphere, and authentic motion.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"resolution": {
"enum": [
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "1080p"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "minimax-hailuo-2.3-pro-t2v"
}
}
}
},
{
"name": "minimax-hailuo-2.3-standard-i2v",
"category": "Image to Video",
"variant": "Standard I2V",
"family": "minimax-2.3",
"group_of": "video",
"description": "Hailuo 2.3 Standard I2V converts still images into visually immersive motion clips with stable dynamics and realistic movement. It provides a balanced mix of quality, speed, and coherence. In 768p video generation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"enum": [
6,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "minimax-hailuo-2.3-standard-i2v"
}
}
}
},
{
"name": "minimax-hailuo-2.3-standard-t2v",
"category": "Text to Video",
"variant": "Standard T2V",
"family": "minimax-2.3",
"group_of": "video",
"description": "Hailuo 2.3 Standard T2V transforms pure imagination into moving cinematic visuals. Simply describe a scene, and this model generates a coherent, high-quality video that captures the prompt\u2019s tone, environment, and emotion. In 768p video generation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"duration": {
"enum": [
6,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "minimax-hailuo-2.3-standard-t2v"
}
}
}
},
{
"name": "minimax-hailuo-2.3-fast",
"category": "Image to Video",
"variant": "Fast I2V",
"family": "minimax-2.3",
"group_of": "video",
"description": "Minimax Hailuo 2.3 Fast is the lightweight, high-speed version of the Hailuo 2.3 family \u2014 designed for creators who need instant video generation with cinematic motion and scene consistency. In 768p video generation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"enum": [
6,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 6
},
"go_fast": {
"type": "boolean",
"title": "Go Fast",
"name": "go_fast",
"description": "Prioritize faster video generation speed with a moderate trade-off in visual quality.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "minimax-hailuo-2.3-fast"
}
}
}
},
{
"name": "kling-v2.5-turbo-std-i2v",
"category": "Image to Video",
"variant": "Standard Image to Video",
"family": "kling-v2.5",
"group_of": "video",
"description": "Kling 2.5 Turbo Std: Top-tier image-to-video generation with unparalleled motion fluidity, cinematic visuals, and exceptional prompt precision.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 10,
"step": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v2.5-turbo-std-i2v"
}
}
}
},
{
"name": "reve-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "reve",
"group_of": "image",
"description": "Generate images from text prompts using reve's vision capabilities. Ideal for basic concept visuals, diagrams, and abstract compositions.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"21:9",
"16:9",
"4:3",
"1:1",
"3:4",
"9:16",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "reve-text-to-image"
}
}
}
},
{
"name": "reve-image-edit",
"category": "Image to Image",
"variant": "Edit Image",
"family": "reve",
"group_of": "image",
"description": "ReVE Edit is a next-generation image editing model that allows users to apply detailed visual transformations through natural language. Whether you want to restyle portraits, modify backgrounds, or create artistic reinterpretations, ReVE Edit delivers realistic and coherent results while preserving structure and identity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to edit.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "reve-image-edit"
}
}
}
},
{
"name": "grok-imagine-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "grok",
"group_of": "video",
"description": "Grok Imagine is xAI\u2019s multimodal image-to-video model, capable of animating still images into short (\u22486 second) cinematic videos with synchronized ambient audio. It focuses on realism, fluid motion, and expressive lighting transitions while maintaining high generation speed.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 1
},
"mode": {
"enum": [
"fun",
"normal",
"spicy"
],
"title": "Mode",
"name": "mode",
"type": "string",
"description": "Note: When generating videos using external image inputs, Spicy mode is not supported and will automatically switch to Normal.",
"default": "normal"
},
"duration": {
"enum": [
6,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 6
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "grok-imagine-image-to-video"
}
}
}
},
{
"name": "grok-imagine-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "grok",
"group_of": "video",
"description": "Grok Imagine is xAI\u2019s fast, creative text-to-video model that generates short (~6-second) cinematic clips with smooth motion, expressive lighting, and ambient audio. It turns a written idea into a visually rich video.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"9:16",
"16:9",
"2:3",
"3:2",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "1:1"
},
"mode": {
"enum": [
"fun",
"normal",
"spicy"
],
"title": "Mode",
"name": "mode",
"type": "string",
"description": "Note: When generating videos using external image inputs, Spicy mode is not supported and will automatically switch to Normal.",
"default": "normal"
},
"duration": {
"enum": [
6,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 6
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "grok-imagine-text-to-video"
}
}
}
},
{
"name": "grok-imagine-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "grok",
"group_of": "image",
"description": "Grok Imagine is xAI\u2019s high-quality image generation model that transforms text prompts into detailed, stylish, and visually expressive images. It excels at creating vivid scenes, characters, environments, and concept art with strong lighting, depth, and artistic clarity. Get 6 images each time.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"9:16",
"16:9",
"2:3",
"3:2",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image. Get 6 images each time.",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "grok-imagine-text-to-image"
}
}
}
},
{
"name": "topaz-image-upscale",
"category": "Image to Image",
"variant": "Image Upscale",
"family": "topaz",
"group_of": "ai-tools",
"description": "Topaz Image Upscale is a high-quality image-to-image enhancement model that increases resolution, sharpness, and detail using AI super-resolution. It improves clarity, restores texture, reduces noise, and produces crisp, high-res output while preserving natural look and fine edges.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"upscale_factor": {
"enum": [
1,
2,
4,
8
],
"description": "Factor to upscale the image by (e.g. 2.0 doubles width and height).",
"type": "string",
"title": "Upscale Factor",
"name": "upscale_factor",
"default": 2
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "topaz-image-upscale"
}
}
}
},
{
"name": "seedvr2-image-upscale",
"category": "Image to Image",
"variant": "Image Upscale",
"family": "seedvr2",
"group_of": "ai-tools",
"description": "SeedVR2 is a one-step diffusion-transformer model designed for image restoration, super-resolution, deblurring, and artifact removal. It enhances low-quality or compressed images into clean, sharp, high-resolution results while preserving natural colors and fine details.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"resolution": {
"enum": [
"2k",
"4k",
"8k"
],
"description": "The target resolution of the generated image.",
"type": "string",
"title": "Resolution",
"name": "resolution",
"default": "4k"
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": "seedvr2-image-upscale"
}
}
}
},
{
"name": "qwen-image-edit-plus-lora",
"category": "Image to Image",
"variant": "Edit Image Plus Lora",
"family": "qwen",
"group_of": "image",
"description": "Qwen-Image-Edit-Plus (2509) is 20B MMDiT image-to-image editor supporting multi-image edits, single-image consistency, and native ControlNet. Ready-to-use REST inference API, best performance, no coldstarts, affordable pricing.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"images_list": {
"description": "Upload or provide image urls.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 3
},
"rotate_right_left": {
"title": "Rotate Right-Left (degrees\u00b0)",
"name": "rotate_right_left",
"type": "int",
"description": "Rotate camera left (positive) or right (negative) in degrees. Positive values rotate left, negative values rotate right.",
"default": 0,
"minValue": -90,
"maxValue": 90,
"step": 1
},
"move_forward": {
"title": "Move Forward \u2192 Close-Up",
"name": "move_forward",
"type": "int",
"description": "Move camera forward (0=no movement, 10=close-up)",
"default": 0,
"minValue": 0,
"maxValue": 10,
"step": 0.1
},
"vertical_angle": {
"title": "Vertical Angle (Bird \u2b04 Worm)",
"name": "vertical_angle",
"type": "int",
"description": "Adjust vertical camera angle (-1=bird's eye view/looking down, 0=neutral, 1=worm's-eye view/looking up)",
"default": 0,
"minValue": -1,
"maxValue": 1,
"step": 0.1
},
"wide_angle_lens": {
"type": "boolean",
"title": "Wide-Angle Lens",
"name": "wide_angle_lens",
"description": "Enable wide-angle lens effect",
"default": false
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"images_list"
],
"endpoint_url": "qwen-image-edit-plus-lora"
}
}
}
},
{
"name": "nano-banana-pro-edit",
"category": "Image to Image",
"variant": "Pro Edit Image",
"family": "nano",
"group_of": "image",
"description": "Nano Banana 2 Edit is the next-generation image editing model developed by Google DeepMind, following the original Nano Banana (also known as Gemini 2.5 Flash Image). It offers advanced image-edit capabilitie with improved resolution.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "List of URLs of input images for editing.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 8
},
"aspect_ratio": {
"enum": [
"1:1",
"3:4",
"4:3",
"9:16",
"16:9",
"3:2",
"2:3",
"5:4",
"4:5",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k",
"4k"
],
"description": "The target resolution of the generated image.",
"type": "string",
"title": "Resolution",
"name": "resolution",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "nano-banana-pro-edit"
}
}
}
},
{
"name": "nano-banana-pro",
"category": "Text to Image",
"variant": "Text to Image Pro",
"family": "nano",
"group_of": "image",
"description": "Nano Banana 2 is the next-generation image generation developed by Google DeepMind, following the original Nano Banana (also known as Gemini 2.5 Flash Image). It offers advanced text-to-image capabilitie with improved resolution.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"3:4",
"4:3",
"9:16",
"16:9",
"3:2",
"2:3",
"5:4",
"4:5",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k",
"4k"
],
"description": "The target resolution of the generated image.",
"type": "string",
"title": "Resolution",
"name": "resolution",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "nano-banana-pro"
}
}
}
},
{
"name": "text-passthrough",
"category": "Text to Text",
"variant": "Text to Text",
"family": "text",
"group_of": "text",
"description": "Text Passthrough model.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"make_input": {
"type": "boolean",
"title": "Make Input",
"name": "make_input",
"description": "",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": ""
}
}
}
},
{
"name": "image-passthrough",
"category": "Image to Image",
"variant": "Image to Image",
"family": "image",
"group_of": "image",
"description": "Image Passthrough model.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"make_input": {
"type": "boolean",
"title": "Make Input",
"name": "make_input",
"description": "",
"default": true
}
},
"title": "BaseInput",
"required": [
"image_url"
],
"endpoint_url": ""
}
}
}
},
{
"name": "video-passthrough",
"category": "Video to Video",
"variant": "Video to Video",
"family": "video",
"group_of": "video",
"description": "Video Passthrough model.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"make_input": {
"type": "boolean",
"title": "Make Input",
"name": "make_input",
"description": "",
"default": true
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": ""
}
}
}
},
{
"name": "kling-o1-text-to-video",
"category": "Text to Video",
"variant": "Text to Video [Pro]",
"family": "kling-o1",
"group_of": "video",
"description": "Kling O1 is a unified, multi-modal video generation engine that transforms natural language prompts into short cinematic video clips. It supports text-to-video generation with realistic motion, dynamic camera moves, and coherent scene rendering.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-o1-text-to-video"
}
}
}
},
{
"name": "kling-o1-image-to-video",
"category": "Image to Video",
"variant": "Image to Video [Pro]",
"family": "kling-o1",
"group_of": "video",
"description": "Kling O1\u2019s Image-to-Video mode transforms one or more reference images into short cinematic video clips by adding natural motion, camera choreography, and scene dynamics while preserving subject identity and visual consistency. It supports start/end frames.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-o1-image-to-video"
}
}
}
},
{
"name": "kling-o1-reference-to-video",
"category": "Image to Video",
"variant": "Reference to Video [Pro]",
"family": "kling-o1",
"group_of": "video",
"description": "Kling O1\u2019s Reference-to-Video mode generates a dynamic video using one or multiple reference images as the visual foundation. It preserves identity, style, composition, and key visual details from the references while adding realistic camera motion, environment dynamics, and scene animation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 7
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 10,
"step": 1
},
"keep_original_sound": {
"type": "boolean",
"title": "Keep Original Sound",
"name": "keep_original_sound",
"description": "Select whether to keep the video original sound through the parameter.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-o1-reference-to-video"
}
}
}
},
{
"name": "kling-o1-video-edit",
"category": "Video to Video",
"variant": "Edit Video [Pro]",
"family": "kling-o1",
"group_of": "video",
"description": "Kling O1 Video Edit lets you send an existing video clip plus an instruction/prompt to edit or transform the clip while preserving temporal coherence and subject identity. Typical edits include color grading, background replacement, object removal, slow-motion slo-mo, speed ramps, style transfer, subtle camera stabilization, and short extension/outro generation. Inputs can include: the source video, an optional frame mask (for localized edits), time range, and style/reference images.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 4
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"keep_original_sound": {
"type": "boolean",
"title": "Keep Original Sound",
"name": "keep_original_sound",
"description": "Select whether to keep the video original sound through the parameter.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "kling-o1-video-edit"
}
}
}
},
{
"name": "kling-o1-video-edit-fast",
"category": "Video to Video",
"variant": "Edit Video Fast [Pro]",
"family": "kling-o1",
"group_of": "video",
"description": "Video Edit Fast is the lightweight, high-speed editing mode of Kling O1. It performs quick edits on an existing video without heavy processing\u2014ideal for fast object replacements, light enhancements, color tweaks, or simple visual adjustments. This mode focuses on speed over complex reconstruction, making it suitable for rapid iterations, previews, and small edits while preserving the original video\u2019s motion and structure.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 4
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"keep_original_sound": {
"type": "boolean",
"title": "Keep Original Sound",
"name": "keep_original_sound",
"description": "Select whether to keep the video original sound through the parameter.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "kling-o1-video-edit-fast"
}
}
}
},
{
"name": "kling-o1-edit-image",
"category": "Image to Image",
"variant": "Edit Image [Pro]",
"family": "kling-o1",
"group_of": "image",
"description": "Kling O1 Image Edit applies targeted transformations to an existing image while preserving composition, lighting, and visual consistency. Use it to replace objects, retouch elements, change materials, or apply stylistic shifts with high fidelity and minimal artifacts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 10
},
"aspect_ratio": {
"enum": [
"auto",
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "kling-o1-edit-image"
}
}
}
},
{
"name": "kling-o1-text-to-image",
"category": "Text to Image",
"variant": "Text to Image [Pro]",
"family": "kling-o1",
"group_of": "image",
"description": "Kling O1 Text-to-Image is a high-fidelity creative image model that converts rich natural-language prompts into ultra-detailed stills. It excels at cinematic composition, realistic lighting, and coherent scene detail\u2014great for concept art, environment renders, character portraits, and stylized imagery with photoreal or illustrative looks.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
},
"num_images": {
"title": "Number of images",
"name": "num_images",
"type": "int",
"description": "Number of images generated in single request. Each number will charge separately",
"default": 1,
"minValue": 1,
"maxValue": 9,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-o1-text-to-image"
}
}
}
},
{
"name": "z-image-turbo",
"category": "Text to Image",
"variant": "Text to Image Turbo",
"family": "z-image",
"group_of": "image",
"description": "Z-Image Turbo is a high-speed text-to-image model optimized for fast creative generation. It produces detailed, high-contrast, high-resolution images with strong stylization control. Ideal for rapid concept creation, visual exploration, product ideas, fantasy scenes, and cinematic composition tests. Designed for low latency and strong prompt adherence.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "z-image-turbo"
}
}
}
},
{
"name": "flux-2-dev",
"category": "Text to Image",
"variant": "Text to Image [Dev]",
"family": "flux-2",
"group_of": "image",
"description": "Flux 2 Dev is a powerful text-to-image diffusion model designed for high-quality, fast, and highly detailed visual generation. It excels at creating cinematic lighting, vibrant compositions, surreal concepts, characters, products, and worlds with strong prompt following and artistic control. Ideal for rapid image ideation, visual storytelling, and concept art.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-2-dev"
}
}
}
},
{
"name": "flux-2-dev-edit",
"category": "Image to Image",
"variant": "Edit Image [Dev]",
"family": "flux-2",
"group_of": "image",
"description": "Flux 2 Dev Edit takes an existing image and applies transformations, replacements, or style changes based on a text instruction. It preserves composition, lighting, and the overall scene while modifying only what the edit prompt specifies. Ideal for creative replacements, stylistic adjustments, object swaps, and environment changes while keeping the original artistic integrity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "List of URLs of input images for editing.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 3
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-2-dev-edit"
}
}
}
},
{
"name": "flux-2-flex",
"category": "Text to Image",
"variant": "Text to Image [Flex]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Flex Text-to-Image is a flexible, high-fidelity generative model capable of producing detailed, imaginative, and stylistically rich scenes from text alone. It excels at surreal concepts, fantasy environments, sci-fi structures, cinematic atmospheres, and high-resolution artistic compositions with strong prompt adherence.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-2-flex"
}
}
}
},
{
"name": "flux-2-flex-edit",
"category": "Image to Image",
"variant": "Edit Image [Flex]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Flex Edit allows flexible transformation of an existing image: object replacement, material changes, lighting adjustments, style shifts, or localized edits. It preserves the original scene\u2019s geometry, perspective, and lighting while modifying only what the edit prompt specifies.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 8
},
"aspect_ratio": {
"enum": [
"auto",
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-2-flex-edit"
}
}
}
},
{
"name": "flux-2-pro",
"category": "Text to Image",
"variant": "Text to Image [Pro]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Pro Text-to-Image is a premium, high-fidelity generative model capable of producing ultra-realistic, cinematic, and deeply detailed images from text prompts. It excels at complex lighting, layered compositions, surreal visual concepts, and professional art-grade rendering suitable for concept art, advertising visuals, and world-building.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-2-pro"
}
}
}
},
{
"name": "flux-2-pro-edit",
"category": "Image to Image",
"variant": "Edit Image [Pro]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Pro Edit enables precise, high-fidelity modifications to an existing image while preserving its lighting, style, mood, and composition. It\u2019s ideal for replacing objects, altering materials, adjusting environmental elements, or performing stylistic transformations without damaging the original scene\u2019s quality. Flux-2-Pro maintains ultra-detailed textures and cinematic realism during edits.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 8
},
"aspect_ratio": {
"enum": [
"auto",
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-2-pro-edit"
}
}
}
},
{
"name": "vidu-q2-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "vidu-q2",
"group_of": "image",
"description": "VIDU Text-to-Image Q2 is a high-quality generative model focused on producing vivid, dynamic, and cinematic still images using natural language prompts. It excels at atmospheric depth, expressive lighting, surreal concepts, and motion-infused compositions typical of VIDU\u2019s visual identity.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k",
"4k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "vidu-q2-text-to-image"
}
}
}
},
{
"name": "vidu-q2-reference-to-image",
"category": "Image to Image",
"variant": "Reference to Image",
"family": "vidu-q2",
"group_of": "image",
"description": "VIDU Reference-to-Image Q2 generates new high-quality images based on one or more reference images. It preserves the key identity, structure, or style of the reference while creating a new scene, variation, or enhanced composition. Ideal for character consistency, object re-interpretation, stylized redesigns, and cinematic recreations guided by reference inputs.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 7
},
"aspect_ratio": {
"enum": [
"auto",
"16:9",
"9:16",
"1:1",
"4:3",
"3:4",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"resolution": {
"enum": [
"1k",
"2k",
"4k"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The target resolution of the generated image.",
"default": "1k"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "vidu-q2-reference-to-image"
}
}
}
},
{
"name": "bytedance-seedream-v4.5",
"category": "Text to Image",
"variant": "Text to Image",
"family": "seedream-v45",
"group_of": "image",
"description": "Seedream-v4.5 is ByteDance\u2019s advanced text-to-image diffusion model designed for generating high-detail, high-contrast, cinematic and stylized images. It excels at surreal fantasy concepts, sci-fi worlds, product visuals, photoreal scenes, and artistic compositions with strong prompt adherence and crisp detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"4:3",
"3:4",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"quality": {
"enum": [
"basic",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "Quality of the output image.",
"default": "basic"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "bytedance-seedream-v4.5"
}
}
}
},
{
"name": "bytedance-seedream-v4.5-edit",
"category": "Image to Image",
"variant": "Edit Image",
"family": "seedream-v45",
"group_of": "image",
"description": "Seedream-v4.5 Edit allows you to transform an existing image using natural-language instructions. It preserves the core composition, lighting, and style of the original while modifying only the requested elements \u2014 perfect for object replacement, environment changes, stylistic adjustments, and high-detail creative reworks.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "List of URLs of input images for editing.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 10
},
"aspect_ratio": {
"enum": [
"1:1",
"16:9",
"9:16",
"4:3",
"3:4",
"2:3",
"3:2",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"quality": {
"enum": [
"basic",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "Quality of the output image.",
"default": "basic"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "bytedance-seedream-v4.5-edit"
}
}
}
},
{
"name": "kling-v2.6-pro-i2v",
"category": "Image to Video",
"variant": "Image to Video",
"family": "kling-v2.6",
"group_of": "video",
"description": "Kling-v2.6-Pro Image-to-Video transforms a single creative image into a short cinematic video. It preserves the original style, lighting, and composition while adding smooth camera motion, atmospheric effects, and dynamic environmental animation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 5
},
"sound": {
"type": "boolean",
"title": "Sound",
"name": "sound",
"description": "Whether sound is generated simultaneously when generating a video.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v2.6-pro-i2v"
}
}
}
},
{
"name": "kling-v2.6-pro-t2v",
"category": "Text to Video",
"variant": "Text to Video",
"family": "kling-v2.6",
"group_of": "video",
"description": "Kling-v2.6-Pro Text-to-Video generates high-fidelity cinematic videos directly from text prompts. It excels at complex compositions, dramatic lighting, fluid camera motion, and visually rich fantasy or sci-fi sequences.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 5
},
"sound": {
"type": "boolean",
"title": "Sound",
"name": "sound",
"description": "Whether sound is generated simultaneously when generating a video.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-v2.6-pro-t2v"
}
}
}
},
{
"name": "pixverse-v5.5-i2v",
"category": "Image to Video",
"variant": "Image to Video",
"family": "pixverse-v5.5",
"group_of": "video",
"description": "PixVerse v5.5 I2V transforms a single image into a dynamic cinematic video clip. It adds smooth camera motion, atmospheric animation, natural parallax, and environmental effects while preserving the image\u2019s original art style and composition.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 2
},
"style": {
"enum": [
"none",
"anime",
"3d_animation",
"clay",
"comic",
"cyberpunk"
],
"title": "Style",
"name": "style",
"type": "string",
"description": "The style of the generated video.",
"default": "none"
},
"thinking": {
"enum": [
"auto",
"enabled",
"disabled"
],
"title": "Thinking",
"name": "thinking",
"type": "string",
"description": "Prompt optimization mode for model decision.",
"default": "auto"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"360p",
"540p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "360p"
},
"duration": {
"enum": [
5,
8,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 5
},
"audio": {
"type": "boolean",
"title": "Audio",
"name": "audio",
"description": "Enable audio generation (BGM, SFX, dialogue).",
"default": false
},
"multi_clip": {
"type": "boolean",
"title": "Multi Clip",
"name": "multi_clip",
"description": "Enable multi-clip generation with dynamic camera changes.",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "pixverse-v5.5-i2v"
}
}
}
},
{
"name": "pixverse-v5.5-t2v",
"category": "Text to Video",
"variant": "Text to Video",
"family": "pixverse-v5.5",
"group_of": "video",
"description": "PixVerse v5.5 T2V generates cinematic short videos directly from text. It excels at stylized fantasy, anime, surreal worlds, atmospheric environments, and fluid camera motion. The model produces vivid lighting, dynamic effects, depth-rich parallax, and smooth motion.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"style": {
"enum": [
"none",
"anime",
"3d_animation",
"clay",
"comic",
"cyberpunk"
],
"title": "Style",
"name": "style",
"type": "string",
"description": "The style of the generated video.",
"default": "none"
},
"thinking": {
"enum": [
"auto",
"enabled",
"disabled"
],
"title": "Thinking",
"name": "thinking",
"type": "string",
"description": "Prompt optimization mode for model decision.",
"default": "auto"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"4:3",
"3:4"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"360p",
"540p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "360p"
},
"duration": {
"enum": [
5,
8,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds.",
"default": 5
},
"audio": {
"type": "boolean",
"title": "Audio",
"name": "audio",
"description": "Enable audio generation (BGM, SFX, dialogue).",
"default": false
},
"multi_clip": {
"type": "boolean",
"title": "Multi Clip",
"name": "multi_clip",
"description": "Enable multi-clip generation with dynamic camera changes.",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "pixverse-v5.5-t2v"
}
}
}
},
{
"name": "kling-v2-avatar-standard",
"category": "Audio to Video",
"variant": "Standard A2V",
"family": "kling-v2",
"group_of": "video",
"description": "AI-Avatar v2 Standard generates a talking-avatar video from a reference image and an audio dialogue. It performs accurate lip-sync, natural facial expressions, subtle head motion, blinking, and light emotional cues based on voice tone. This Standard version focuses on speed and natural realism.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
}
},
"title": "BaseInput",
"required": [
"image_url",
"audio_url"
],
"endpoint_url": "kling-v2-avatar-standard"
}
}
}
},
{
"name": "kling-v2-avatar-pro",
"category": "Audio to Video",
"variant": "Pro A2V",
"family": "kling-v2",
"group_of": "video",
"description": "AI-Avatar v2 Pro takes a reference image of a person/character and an audio dialogue clip, then generates a realistic talking-avatar video. It preserves identity, lip syncs accurately to the audio, adds natural head movement, eye motion, expressions, and cinematic lighting.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
}
},
"title": "BaseInput",
"required": [
"image_url",
"audio_url"
],
"endpoint_url": "kling-v2-avatar-pro"
}
}
}
},
{
"name": "wan2.2-spicy-image-to-video",
"category": "Image to Video",
"variant": "Spicy Image to Video",
"family": "wan2.2",
"group_of": "video",
"description": "Wan2.2-spicy Image-to-Video transforms a single creative image into a short dynamic video with bold motion, stylized effects, high-contrast lighting, and energy-driven animations. The \u201cspicy\u201d variant produces more dramatic movement, more vivid colors, and more expressive visual effects.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"enum": [
5,
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "wan2.2-spicy-image-to-video"
}
}
}
},
{
"name": "wan2.2-spicy-video-extend",
"category": "Video to Video",
"variant": "Spicy Video Extend",
"family": "wan2.2",
"group_of": "video",
"description": "Wan-2.2-spicy Video Extend continues an existing video by generating new frames that match the original style but add stronger motion, bolder effects, and spicier dramatics.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "480p"
},
"duration": {
"enum": [
5,
8
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "wan2.2-spicy-video-extend"
}
}
}
},
{
"name": "minimax-voice-clone",
"category": "Text to Audio",
"variant": "Voice Clone",
"family": "minimax-2.3",
"group_of": "music",
"description": "Minimax Voice Clone creates a high-fidelity digital clone of a speaker\u2019s voice from a short reference audio sample. It reproduces the speaker\u2019s tone, emotion, accent, rhythm, and speaking style, then generates new speech from any text input.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"audio_url": {
"description": "Url of the audio url.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"custom_voice_id": {
"description": "Custom user-defined ID. Minimum 8 characters must include letters and numbers and start with a letter. Duplicate voice-ids will throw an error.",
"format": "text",
"type": "string",
"title": "Custom Voice ID",
"name": "custom_voice_id",
"placeholder": "sf02174c-5f5d-46e6-8758-7544128c27b2"
},
"model": {
"enum": [
"speech-02-hd",
"speech-02-turbo",
"speech-2.5-hd-preview",
"speech-2.5-turbo-preview",
"speech-2.6-hd",
"speech-2.6-turbo"
],
"title": "Model",
"name": "model",
"type": "string",
"description": "Specify the TTS model to be used for the preview. This is only a preview after cloning. Once the model is generated, any Minimax Turbo or HD voice model can be used for inference.",
"default": "speech-02-hd"
},
"need_noise_reduction": {
"type": "boolean",
"title": "Need Noise Reduction",
"name": "need_noise_reduction",
"description": "Enable noise reduction. Default is false (no noise reduction).",
"default": false
},
"need_volume_normalization": {
"type": "boolean",
"title": "Need Volume Normalization",
"name": "need_volume_normalization",
"description": "Specify whether to enable volume normalization.",
"default": false
},
"accuracy": {
"title": "Accuracy",
"name": "accuracy",
"type": "int",
"description": "Text validation accuracy threshold, with a value range of [0, 1].",
"default": 0.7,
"minValue": 0,
"maxValue": 1,
"step": 0.01
},
"prompt": {
"description": "Text for audio preview. Limited to 2000 characters.",
"type": "string",
"title": "Prompt",
"name": "prompt"
}
},
"title": "BaseInput",
"required": [
"audio_url",
"custom_voice_id"
],
"endpoint_url": "minimax-voice-clone"
}
}
}
},
{
"name": "minimax-speech-2.6-hd",
"category": "Text to Audio",
"variant": "Speech HD",
"family": "minimax-2.6",
"group_of": "music",
"description": "Speech-2.6-hd is Minimax\u2019s high-definition text-to-speech model that turns written text into natural, human-like audio. It produces studio-quality speech with clear pronunciation, smooth pacing, realistic emotion, and no background noise.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text to convert to speech. Every character is 1 token. Maximum 10000 characters. Use <#x#> between words to control pause duration (0.01-99.99s).",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"voice_id": {
"enum": [
"Wise_Woman",
"Friendly_Person",
"Inspirational_girl",
"Deep_Voice_Man",
"Calm_Woman",
"Casual_Guy",
"Lively_Girl",
"Patient_Man",
"Young_Knight",
"Determined_Man",
"Lovely_Girl",
"Decent_Boy",
"Imposing_Manner",
"Elegant_Man",
"Abbess",
"Sweet_Girl_2",
"Exuberant_Girl",
"English_expressive_narrator",
"English_radiant_girl",
"English_magnetic_voiced_man",
"English_compelling_lady1",
"English_Aussie_Bloke",
"English_captivating_female1",
"English_Upbeat_Woman",
"English_Trustworth_Man",
"English_CalmWoman",
"English_UpsetGirl",
"English_Gentle-voiced_man",
"English_Whispering_girl_v3",
"English_Diligent_Man",
"English_Graceful_Lady",
"English_Husky_MetalHead",
"English_ReservedYoungMan",
"Thai_female_1_sample1",
"Thai_female_2_sample2",
"English_PlayfulGirl",
"English_ManWithDeepVoice",
"English_GentleTeacher",
"English_MaturePartner",
"English_FriendlyPerson",
"English_MatureBoss",
"English_Debator",
"whisper_man",
"English_Abbess",
"English_LovelyGirl",
"whisper_woman_1",
"English_Steadymentor",
"English_Deep-VoicedGentleman",
"English_DeterminedMan",
"English_Wiselady",
"English_CaptivatingStoryteller",
"English_AttractiveGirl",
"English_DecentYoungMan",
"English_SentimentalLady",
"English_ImposingManner",
"English_SadTeen",
"English_ThoughtfulMan",
"English_PassionateWarrior",
"English_DecentBoy",
"English_WiseScholar",
"English_Soft-spokenGirl",
"English_SereneWoman",
"English_ConfidentWoman",
"English_PatientMan",
"English_Comedian",
"English_GorgeousLady",
"English_BossyLeader",
"English_LovelyLady",
"English_Strong-WilledBoy",
"English_Deep-tonedMan",
"English_StressedLady",
"English_AssertiveQueen",
"English_AnimeCharacter",
"Portuguese_Optimisticyouth",
"Portuguese_CuteElf",
"English_Jovialman",
"English_WhimsicalGirl",
"English_CharmingQueen",
"English_Kind-heartedGirl",
"English_FriendlyNeighbor",
"English_Sweet_Female_4",
"English_Magnetic_Male_2",
"English_Lively_Male_11",
"English_Friendly_Female_3",
"English_Steady_Female_1",
"English_Lively_Male_10",
"English_Magnetic_Male_12",
"English_Steady_Female_5",
"English_Insightful_Speaker",
"English_patient_man_v1",
"English_Persuasive_Man",
"English_Explanatory_Man",
"English_intellect_female_1",
"English_Cute_Girl",
"English_Sharp_Commentator",
"English_Honest_Man",
"angry_pirate_1",
"massive_kind_troll",
"movie_trailer_deep",
"peace_and_ease",
"moss_audio_6dc281eb-713c-11f0-a447-9613c873494c",
"moss_audio_c12a59b9-7115-11f0-a447-9613c873494c",
"moss_audio_076697ad-7144-11f0-a447-9613c873494c",
"moss_audio_737a299c-734a-11f0-918f-4e0486034804",
"moss_audio_19dbb103-7350-11f0-ad20-f2bc95e89150",
"moss_audio_7c7e7ae2-7356-11f0-9540-7ef9b4b62566",
"moss_audio_570551b1-735c-11f0-b236-0adeeecad052",
"conversational_female_1_v1",
"conversational_female_2_v1",
"socialmedia_female_1_v1",
"BritishChild_male_1_v1",
"BritishChild_female_1_v1",
"Chinese (Mandarin)_Reliable_Executive",
"Chinese (Mandarin)_News_Anchor",
"Chinese (Mandarin)_Unrestrained_Young_Man",
"Chinese (Mandarin)_Mature_Woman",
"Arrogant_Miss",
"Chinese (Mandarin)_Kind-hearted_Antie",
"Robot_Armor",
"hunyin_6",
"Chinese (Mandarin)_HK_Flight_Attendant",
"Chinese (Mandarin)_Humorous_Elder",
"Chinese (Mandarin)_Gentleman",
"Chinese (Mandarin)_Warm_Bestie",
"Chinese (Mandarin)_Southern_Young_Man",
"Chinese (Mandarin)_Wise_Women",
"moss_audio_cedfd4d2-736d-11f0-99be-fe40dd2a5fe8",
"moss_audio_a0d611da-737c-11f0-ad20-f2bc95e89150",
"moss_audio_4f4172f4-737b-11f0-9540-7ef9b4b62566",
"moss_audio_62ca20b0-7380-11f0-99be-fe40dd2a5fe8",
"Portuguese_PowerfulSoldier",
"Portuguese_FascinatingBoy",
"Portuguese_RomanticHusband",
"Portuguese_StrictBoss",
"Chinese (Mandarin)_Stubborn_Friend",
"Chinese (Mandarin)_Sweet_Lady",
"moss_audio_ad5baf92-735f-11f0-8263-fe5a2fe98ec8",
"Chinese (Mandarin)_Gentle_Youth",
"Chinese (Mandarin)_Warm_Girl",
"Chinese (Mandarin)_Male_Announcer",
"Chinese (Mandarin)_Kind-hearted_Elder",
"Chinese (Mandarin)_Cute_Spirit",
"Chinese (Mandarin)_Radio_Host",
"Chinese (Mandarin)_Lyrical_Voice",
"Chinese (Mandarin)_Straightforward_Boy",
"Chinese (Mandarin)_Sincere_Adult",
"Chinese (Mandarin)_Gentle_Senior",
"Chinese (Mandarin)_Crisp_Girl",
"Chinese (Mandarin)_Pure-hearted_Boy",
"Chinese (Mandarin)_Soft_Girl",
"Chinese (Mandarin)_IntellectualGirl",
"Chinese (Mandarin)_Laid_BackGirl",
"Chinese (Mandarin)_ExplorativeGirl",
"Chinese (Mandarin)_Warm-HeartedAunt",
"Chinese (Mandarin)_BashfulGirl",
"Arabic_CalmWoman",
"Arabic_FriendlyGuy",
"Cantonese_ProfessionalHost\uff08F)",
"Cantonese_GentleLady",
"Cantonese_ProfessionalHost\uff08M)",
"Cantonese_PlayfulMan",
"Cantonese_CuteGirl",
"Cantonese_KindWoman",
"Cantonese_Narrator",
"Cantonese_WiselProfessor",
"Cantonese_IndifferentStaff",
"Japanese_ColdQueen",
"Japanese_DependableWoman",
"Japanese_GentleButler",
"Japanese_KindLady",
"Dutch_kindhearted_girl",
"Dutch_bossy_leader",
"French_Male_Speech_New",
"French_Female_News Anchor",
"French_CasualMan",
"French_MovieLeadFemale",
"French_FemaleAnchor",
"French_MaleNarrator",
"French_Female Journalist",
"French_Female_Speech_New",
"German_FriendlyMan",
"German_SweetLady",
"German_PlayfulMan",
"Indonesian_SweetGirl",
"Indonesian_ReservedYoungMan",
"Indonesian_CharmingGirl",
"Russian_AmbitiousWoman",
"Russian_ReliableMan",
"Russian_CrazyQueen",
"Russian_PessimisticGirl",
"Indonesian_CalmWoman",
"Indonesian_ConfidentWoman",
"Indonesian_CaringMan",
"Indonesian_BossyLeader",
"Indonesian_DeterminedBoy",
"Indonesian_GentleGirl",
"Italian_BraveHeroine",
"Italian_Narrator",
"Italian_WanderingSorcerer",
"Italian_DiligentLeader",
"Italian_ReliableMan",
"Italian_AthleticStudent",
"Italian_ArrogantPrincess",
"Japanese_Whisper_Belle",
"Japanese_IntellectualSenior",
"Japanese_DecisivePrincess",
"Japanese_LoyalKnight",
"Japanese_DominantMan",
"Japanese_SeriousCommander",
"Japanese_CalmLady",
"Japanese_OptimisticYouth",
"Japanese_GenerousIzakayaOwner",
"Japanese_SportyStudent",
"Japanese_InnocentBoy",
"Japanese_GracefulMaiden",
"Korean_PowerfulGirl",
"Korean_BossyMan",
"Korean_SweetGirl",
"Korean_CheerfulBoyfriend",
"Korean_EnchantingSister",
"Korean_ShyGirl",
"Korean_ReliableSister",
"Korean_StrictBoss",
"Korean_SassyGirl",
"Korean_ChildhoodFriendGirl",
"Korean_PlayboyCharmer",
"Korean_ElegantPrincess",
"English_energetic_male_1",
"English_witty_female_1",
"English_Lucky_Robot",
"Korean_BraveFemaleWarrior",
"Korean_BraveYouth",
"Korean_CalmLady",
"Korean_EnthusiasticTeen",
"Korean_SoothingLady",
"Korean_IntellectualSenior",
"Korean_LonelyWarrior",
"Korean_MatureLady",
"Korean_InnocentBoy",
"Korean_CharmingSister",
"Korean_AthleticStudent",
"Korean_BraveAdventurer",
"Korean_CalmGentleman",
"Korean_WiseElf",
"Korean_CheerfulCoolJunior",
"Korean_DecisiveQueen",
"Korean_ColdYoungMan",
"Korean_MysteriousGirl",
"Korean_QuirkyGirl",
"Korean_ConsiderateSenior",
"Chinese (Mandarin)_Warm_HeartedGirl",
"Korean_CheerfulLittleSister",
"Korean_DominantMan",
"Korean_AirheadedGirl",
"Korean_ReliableYouth",
"Korean_FriendlyBigSister",
"Korean_GentleBoss",
"Korean_ColdGirl",
"Korean_HaughtyLady",
"Korean_CharmingElderSister",
"Korean_IntellectualMan",
"Korean_CaringWoman",
"Korean_WiseTeacher",
"Korean_ConfidentBoss",
"Korean_AthleticGirl",
"Korean_PossessiveMan",
"Korean_GentleWoman",
"Korean_CockyGuy",
"Korean_ThoughtfulWoman",
"Korean_OptimisticYouth",
"Portuguese_AnxiousMan",
"Portuguese_Matureresearcher",
"Portuguese_EnergeticGirl",
"Portuguese_FunnyGuy",
"Portuguese_Nuttylady",
"Portuguese_Deep-tonedMan",
"Portuguese_SentimentalLady",
"Portuguese_BossyLeader",
"Portuguese_Wiselady",
"Portuguese_Strong-WilledBoy",
"Portuguese_Deep-VoicedGentleman",
"Portuguese_UpsetGirl",
"Portuguese_PassionateWarrior",
"Portuguese_AnimeCharacter",
"Portuguese_ConfidentWoman",
"Portuguese_AngryMan",
"Portuguese_CaptivatingStoryteller",
"Portuguese_Godfather",
"Portuguese_ReservedYoungMan",
"Portuguese_SmartYoungGirl",
"Portuguese_Kind-heartedGirl",
"Portuguese_Pompouslady",
"Portuguese_Grinch",
"Portuguese_Debator",
"Portuguese_SweetGirl",
"Portuguese_AttractiveGirl",
"Portuguese_ThoughtfulMan",
"Portuguese_PlayfulGirl",
"Portuguese_GorgeousLady",
"Portuguese_LovelyLady",
"Portuguese_SereneWoman",
"Portuguese_SadTeen",
"Portuguese_MaturePartner",
"Portuguese_Comedian",
"Portuguese_NaughtySchoolgirl",
"Portuguese_Narrator",
"Portuguese_ToughBoss",
"Portuguese_Fussyhostess",
"Portuguese_Dramatist",
"Portuguese_Steadymentor",
"Portuguese_Jovialman",
"Portuguese_CharmingQueen",
"Portuguese_SantaClaus",
"Portuguese_Rudolph",
"Portuguese_Arnold",
"Portuguese_CharmingSanta",
"Portuguese_Ghost",
"Portuguese_HumorousElder",
"Portuguese_CalmLeader",
"Portuguese_GentleTeacher",
"Portuguese_EnergeticBoy",
"Portuguese_ReliableMan",
"Portuguese_SereneElder",
"Portuguese_GrimReaper",
"Portuguese_AssertiveQueen",
"Portuguese_WhimsicalGirl",
"Portuguese_StressedLady",
"Portuguese_FriendlyNeighbor",
"Portuguese_CaringGirlfriend",
"Portuguese_InspiringLady",
"Portuguese_PlayfulSpirit",
"Portuguese_ElegantGirl",
"Portuguese_CompellingGirl",
"Portuguese_PowerfulVeteran",
"Portuguese_SensibleManager",
"Portuguese_ThoughtfulLady",
"Portuguese_TheatricalActor",
"Portuguese_FragileBoy",
"Portuguese_ChattyGirl",
"Portuguese_Conscientiousinstructor",
"Portuguese_RationalMan",
"Portuguese_WiseScholar",
"Portuguese_FrankLady",
"Portuguese_DeterminedManager",
"Portuguese_CharmingLady",
"Russian_HandsomeChildhoodFriend",
"Russian_BrightHeroine",
"Russian_AttractiveGuy",
"Russian_Bad-temperedBoy",
"Spanish_FriendlyNeighbor",
"Spanish_FragileBoy",
"Spanish_UpsetGirl",
"Spanish_Soft-spokenGirl",
"Spanish_CharmingQueen",
"Spanish_Nuttylady",
"Spanish_ElegantGirl",
"Spanish_FascinatingBoy",
"Spanish_FunnyGuy",
"Spanish_PlayfulSpirit",
"Spanish_TheatricalActor",
"Spanish_SereneWoman",
"Spanish_MaturePartner",
"Spanish_CaptivatingStoryteller",
"Spanish_Narrator",
"Spanish_WiseScholar",
"Spanish_Kind-heartedGirl",
"Spanish_DeterminedManager",
"Spanish_BossyLeader",
"Spanish_ReservedYoungMan",
"Spanish_ConfidentWoman",
"Spanish_ThoughtfulMan",
"Spanish_Strong-WilledBoy",
"Spanish_SophisticatedLady",
"Spanish_RationalMan",
"Spanish_AnimeCharacter",
"Spanish_Deep-tonedMan",
"Spanish_Fussyhostess",
"Spanish_SincereTeen",
"Spanish_FrankLady",
"Spanish_Comedian",
"Spanish_Debator",
"Spanish_ToughBoss",
"Spanish_Wiselady",
"Spanish_Steadymentor",
"finnish_male_1_v2",
"hindi_male_1_v2",
"hindi_female_2_v1",
"hindi_female_1_v2",
"Spanish_Jovialman",
"Spanish_SantaClaus",
"Spanish_Rudolph",
"Spanish_Intonategirl",
"Spanish_Arnold",
"Spanish_Ghost",
"Spanish_HumorousElder",
"Spanish_EnergeticBoy",
"Spanish_WhimsicalGirl",
"Spanish_StrictBoss",
"Spanish_ReliableMan",
"Spanish_SereneElder",
"Spanish_AngryMan",
"Spanish_AssertiveQueen",
"Spanish_CaringGirlfriend",
"Spanish_PowerfulSoldier",
"Spanish_PassionateWarrior",
"Spanish_ChattyGirl",
"Spanish_RomanticHusband",
"Spanish_CompellingGirl",
"Spanish_PowerfulVeteran",
"Spanish_SensibleManager",
"Spanish_ThoughtfulLady",
"Turkish_CalmWoman",
"Turkish_Trustworthyman",
"Ukrainian_CalmWoman",
"Ukrainian_WiseScholar",
"Vietnamese_Serene_Man",
"Vietnamese_female_4_v1",
"Vietnamese_male_1_v2",
"Vietnamese_kindhearted_girl",
"Thai_Optimistic_girl",
"Thai_male_1_sample8",
"Thai_Tender_Woman",
"Thai_male_2_sample2",
"Polish_male_1_sample4",
"Polish_male_2_sample3",
"Polish_female_1_sample1",
"Polish_female_2_sample3",
"Romanian_male_1_sample2",
"Romanian_male_2_sample1",
"Romanian_female_1_sample4",
"Romanian_female_2_sample1",
"Greek_female_1_sample1",
"greek_male_1a_v1",
"Greek_female_2_sample3",
"czech_male_1_v1",
"czech_female_5_v7",
"czech_female_2_v2",
"finnish_male_3_v1",
"finnish_female_4_v1",
"Bulgarian_male_2_v1",
"Bulgarian_female_1_v1",
"Danish_male_1_v1",
"Danish_female_1_v1",
"Hebrew_male_1_v1",
"Hebrew_female_1_v1",
"Malay_male_1_v1",
"Malay_female_1_v1",
"Malay_female_2_v1",
"Persian_male_1_v1",
"Persian_female_1_v1",
"Slovak_male_1_v1",
"Slovak_female_1_v1",
"Swedish_male_1_v1",
"Swedish_female_1_v1",
"Croatian_male_1_v1",
"Croatian_female_1_v1",
"Filipino_male_1_v1",
"Filipino_female_1_v1",
"Hungarian_male_1_v1",
"Hungarian_female_1_v1",
"Norwegian_male_1_v1",
"Norwegian_female_1_v1",
"Slovenian_male_1_v1",
"Slovenian_female_1_v2",
"Catalan_male_1_v1",
"Catalan_female_1_v1",
"Nynorsk_male_1_v1",
"Nynorsk_female_1_v1",
"Tamil_male_1_v1",
"Tamil_female_1_v1",
"Afrikaans_male_1_v1",
"Afrikaans_female_1_v1"
],
"description": "Desired voice ID. Use a voice ID you have trained (https://muapi.ai/playground/minimax-voice-clone), or one of the following system voice IDs",
"type": "string",
"typing": true,
"title": "Voice ID",
"name": "voice_id",
"default": "Friendly_Person"
},
"speed": {
"title": "Speed",
"name": "speed",
"type": "int",
"description": "Speech speed. Range: 0.5-2.0, where 1.0 is normal speed.",
"default": 1,
"minValue": 0.5,
"maxValue": 2,
"step": 0.01
},
"volume": {
"title": "Volume",
"name": "volume",
"type": "int",
"description": "Speech volume. Range: 0.1-10.0, where 1.0 is normal volume.",
"default": 1,
"minValue": 0.1,
"maxValue": 10,
"step": 0.01
},
"pitch": {
"title": "Pitch",
"name": "pitch",
"type": "int",
"description": "Speech pitch. Range: -12 to 12, where 0 is normal pitch.",
"default": 0,
"minValue": -12,
"maxValue": 12,
"step": 1
},
"emotion": {
"enum": [
"happy",
"sad",
"angry",
"fearful",
"disgusted",
"surprised",
"neutral"
],
"title": "Emotion",
"name": "emotion",
"type": "string",
"description": "The emotion of the generated speech.",
"default": "happy"
},
"english_normalization": {
"type": "boolean",
"title": "English Normalization",
"name": "english_normalization",
"description": "This parameter supports English text normalization, which improves performance in number-reading scenarios.",
"default": false
},
"sample_rate": {
"enum": [
8000,
16000,
22050,
24000,
32000,
44100
],
"type": "integer",
"title": "Sample Rate",
"name": "sample_rate",
"description": "Sample rate of generated sound.",
"default": 8000
},
"bitrate": {
"enum": [
32000,
64000,
128000,
256000
],
"type": "integer",
"title": "Bitrate",
"name": "bitrate",
"description": "Bitrate of generated sound.",
"default": 32000
},
"channel": {
"enum": [
1,
2
],
"type": "integer",
"title": "Channel",
"name": "channel",
"description": "he number of channels of the generated audio. 1: mono, 2: stereo.",
"default": 1
},
"format": {
"enum": [
"mp3",
"wav",
"pcm",
"flac"
],
"type": "string",
"title": "Format",
"name": "format",
"description": "Format of generated sound.",
"default": "mp3"
},
"language_boost": {
"enum": [
"Chinese",
"Chinese,Yue",
"English",
"Arabic",
"Russian",
"Spanish",
"French",
"Portuguese",
"German",
"Turkish",
"Dutch",
"Ukrainian",
"Vietnamese",
"Indonesian",
"Japanese",
"Italian",
"Korean",
"Thai",
"Polish",
"Romanian",
"Greek",
"Czech",
"Finnish",
"Hindi",
"Bulgarian",
"Danish",
"Hebrew",
"Malay",
"Persian",
"Slovak",
"Swedish",
"Croatian",
"Filipino",
"Hungarian",
"Norwegian",
"Slovenian",
"Catalan",
"Nynorsk",
"Tamil",
"Afrikaans",
"auto"
],
"title": "Language Boost",
"name": "language_boost",
"type": "string",
"description": "Enhance the ability to recognize specified languages and dialects.",
"default": "auto"
}
},
"title": "BaseInput",
"required": [
"prompt",
"voice_id"
],
"endpoint_url": "minimax-speech-2.6-hd"
}
}
}
},
{
"name": "minimax-speech-2.6-turbo",
"category": "Text to Audio",
"variant": "Speech Turbo",
"family": "minimax-2.6",
"group_of": "music",
"description": "Speech-2.6-turbo is Minimax\u2019s fast, lightweight text-to-speech model designed for quick audio generation while maintaining good natural voice quality. It produces clear speech with smooth pacing and minimal delay.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text to convert to speech. Every character is 1 token. Maximum 10000 characters. Use <#x#> between words to control pause duration (0.01-99.99s).",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"voice_id": {
"enum": [
"Wise_Woman",
"Friendly_Person",
"Inspirational_girl",
"Deep_Voice_Man",
"Calm_Woman",
"Casual_Guy",
"Lively_Girl",
"Patient_Man",
"Young_Knight",
"Determined_Man",
"Lovely_Girl",
"Decent_Boy",
"Imposing_Manner",
"Elegant_Man",
"Abbess",
"Sweet_Girl_2",
"Exuberant_Girl",
"English_expressive_narrator",
"English_radiant_girl",
"English_magnetic_voiced_man",
"English_compelling_lady1",
"English_Aussie_Bloke",
"English_captivating_female1",
"English_Upbeat_Woman",
"English_Trustworth_Man",
"English_CalmWoman",
"English_UpsetGirl",
"English_Gentle-voiced_man",
"English_Whispering_girl_v3",
"English_Diligent_Man",
"English_Graceful_Lady",
"English_Husky_MetalHead",
"English_ReservedYoungMan",
"Thai_female_1_sample1",
"Thai_female_2_sample2",
"English_PlayfulGirl",
"English_ManWithDeepVoice",
"English_GentleTeacher",
"English_MaturePartner",
"English_FriendlyPerson",
"English_MatureBoss",
"English_Debator",
"whisper_man",
"English_Abbess",
"English_LovelyGirl",
"whisper_woman_1",
"English_Steadymentor",
"English_Deep-VoicedGentleman",
"English_DeterminedMan",
"English_Wiselady",
"English_CaptivatingStoryteller",
"English_AttractiveGirl",
"English_DecentYoungMan",
"English_SentimentalLady",
"English_ImposingManner",
"English_SadTeen",
"English_ThoughtfulMan",
"English_PassionateWarrior",
"English_DecentBoy",
"English_WiseScholar",
"English_Soft-spokenGirl",
"English_SereneWoman",
"English_ConfidentWoman",
"English_PatientMan",
"English_Comedian",
"English_GorgeousLady",
"English_BossyLeader",
"English_LovelyLady",
"English_Strong-WilledBoy",
"English_Deep-tonedMan",
"English_StressedLady",
"English_AssertiveQueen",
"English_AnimeCharacter",
"Portuguese_Optimisticyouth",
"Portuguese_CuteElf",
"English_Jovialman",
"English_WhimsicalGirl",
"English_CharmingQueen",
"English_Kind-heartedGirl",
"English_FriendlyNeighbor",
"English_Sweet_Female_4",
"English_Magnetic_Male_2",
"English_Lively_Male_11",
"English_Friendly_Female_3",
"English_Steady_Female_1",
"English_Lively_Male_10",
"English_Magnetic_Male_12",
"English_Steady_Female_5",
"English_Insightful_Speaker",
"English_patient_man_v1",
"English_Persuasive_Man",
"English_Explanatory_Man",
"English_intellect_female_1",
"English_Cute_Girl",
"English_Sharp_Commentator",
"English_Honest_Man",
"angry_pirate_1",
"massive_kind_troll",
"movie_trailer_deep",
"peace_and_ease",
"moss_audio_6dc281eb-713c-11f0-a447-9613c873494c",
"moss_audio_c12a59b9-7115-11f0-a447-9613c873494c",
"moss_audio_076697ad-7144-11f0-a447-9613c873494c",
"moss_audio_737a299c-734a-11f0-918f-4e0486034804",
"moss_audio_19dbb103-7350-11f0-ad20-f2bc95e89150",
"moss_audio_7c7e7ae2-7356-11f0-9540-7ef9b4b62566",
"moss_audio_570551b1-735c-11f0-b236-0adeeecad052",
"conversational_female_1_v1",
"conversational_female_2_v1",
"socialmedia_female_1_v1",
"BritishChild_male_1_v1",
"BritishChild_female_1_v1",
"Chinese (Mandarin)_Reliable_Executive",
"Chinese (Mandarin)_News_Anchor",
"Chinese (Mandarin)_Unrestrained_Young_Man",
"Chinese (Mandarin)_Mature_Woman",
"Arrogant_Miss",
"Chinese (Mandarin)_Kind-hearted_Antie",
"Robot_Armor",
"hunyin_6",
"Chinese (Mandarin)_HK_Flight_Attendant",
"Chinese (Mandarin)_Humorous_Elder",
"Chinese (Mandarin)_Gentleman",
"Chinese (Mandarin)_Warm_Bestie",
"Chinese (Mandarin)_Southern_Young_Man",
"Chinese (Mandarin)_Wise_Women",
"moss_audio_cedfd4d2-736d-11f0-99be-fe40dd2a5fe8",
"moss_audio_a0d611da-737c-11f0-ad20-f2bc95e89150",
"moss_audio_4f4172f4-737b-11f0-9540-7ef9b4b62566",
"moss_audio_62ca20b0-7380-11f0-99be-fe40dd2a5fe8",
"Portuguese_PowerfulSoldier",
"Portuguese_FascinatingBoy",
"Portuguese_RomanticHusband",
"Portuguese_StrictBoss",
"Chinese (Mandarin)_Stubborn_Friend",
"Chinese (Mandarin)_Sweet_Lady",
"moss_audio_ad5baf92-735f-11f0-8263-fe5a2fe98ec8",
"Chinese (Mandarin)_Gentle_Youth",
"Chinese (Mandarin)_Warm_Girl",
"Chinese (Mandarin)_Male_Announcer",
"Chinese (Mandarin)_Kind-hearted_Elder",
"Chinese (Mandarin)_Cute_Spirit",
"Chinese (Mandarin)_Radio_Host",
"Chinese (Mandarin)_Lyrical_Voice",
"Chinese (Mandarin)_Straightforward_Boy",
"Chinese (Mandarin)_Sincere_Adult",
"Chinese (Mandarin)_Gentle_Senior",
"Chinese (Mandarin)_Crisp_Girl",
"Chinese (Mandarin)_Pure-hearted_Boy",
"Chinese (Mandarin)_Soft_Girl",
"Chinese (Mandarin)_IntellectualGirl",
"Chinese (Mandarin)_Laid_BackGirl",
"Chinese (Mandarin)_ExplorativeGirl",
"Chinese (Mandarin)_Warm-HeartedAunt",
"Chinese (Mandarin)_BashfulGirl",
"Arabic_CalmWoman",
"Arabic_FriendlyGuy",
"Cantonese_ProfessionalHost\uff08F)",
"Cantonese_GentleLady",
"Cantonese_ProfessionalHost\uff08M)",
"Cantonese_PlayfulMan",
"Cantonese_CuteGirl",
"Cantonese_KindWoman",
"Cantonese_Narrator",
"Cantonese_WiselProfessor",
"Cantonese_IndifferentStaff",
"Japanese_ColdQueen",
"Japanese_DependableWoman",
"Japanese_GentleButler",
"Japanese_KindLady",
"Dutch_kindhearted_girl",
"Dutch_bossy_leader",
"French_Male_Speech_New",
"French_Female_News Anchor",
"French_CasualMan",
"French_MovieLeadFemale",
"French_FemaleAnchor",
"French_MaleNarrator",
"French_Female Journalist",
"French_Female_Speech_New",
"German_FriendlyMan",
"German_SweetLady",
"German_PlayfulMan",
"Indonesian_SweetGirl",
"Indonesian_ReservedYoungMan",
"Indonesian_CharmingGirl",
"Russian_AmbitiousWoman",
"Russian_ReliableMan",
"Russian_CrazyQueen",
"Russian_PessimisticGirl",
"Indonesian_CalmWoman",
"Indonesian_ConfidentWoman",
"Indonesian_CaringMan",
"Indonesian_BossyLeader",
"Indonesian_DeterminedBoy",
"Indonesian_GentleGirl",
"Italian_BraveHeroine",
"Italian_Narrator",
"Italian_WanderingSorcerer",
"Italian_DiligentLeader",
"Italian_ReliableMan",
"Italian_AthleticStudent",
"Italian_ArrogantPrincess",
"Japanese_Whisper_Belle",
"Japanese_IntellectualSenior",
"Japanese_DecisivePrincess",
"Japanese_LoyalKnight",
"Japanese_DominantMan",
"Japanese_SeriousCommander",
"Japanese_CalmLady",
"Japanese_OptimisticYouth",
"Japanese_GenerousIzakayaOwner",
"Japanese_SportyStudent",
"Japanese_InnocentBoy",
"Japanese_GracefulMaiden",
"Korean_PowerfulGirl",
"Korean_BossyMan",
"Korean_SweetGirl",
"Korean_CheerfulBoyfriend",
"Korean_EnchantingSister",
"Korean_ShyGirl",
"Korean_ReliableSister",
"Korean_StrictBoss",
"Korean_SassyGirl",
"Korean_ChildhoodFriendGirl",
"Korean_PlayboyCharmer",
"Korean_ElegantPrincess",
"English_energetic_male_1",
"English_witty_female_1",
"English_Lucky_Robot",
"Korean_BraveFemaleWarrior",
"Korean_BraveYouth",
"Korean_CalmLady",
"Korean_EnthusiasticTeen",
"Korean_SoothingLady",
"Korean_IntellectualSenior",
"Korean_LonelyWarrior",
"Korean_MatureLady",
"Korean_InnocentBoy",
"Korean_CharmingSister",
"Korean_AthleticStudent",
"Korean_BraveAdventurer",
"Korean_CalmGentleman",
"Korean_WiseElf",
"Korean_CheerfulCoolJunior",
"Korean_DecisiveQueen",
"Korean_ColdYoungMan",
"Korean_MysteriousGirl",
"Korean_QuirkyGirl",
"Korean_ConsiderateSenior",
"Chinese (Mandarin)_Warm_HeartedGirl",
"Korean_CheerfulLittleSister",
"Korean_DominantMan",
"Korean_AirheadedGirl",
"Korean_ReliableYouth",
"Korean_FriendlyBigSister",
"Korean_GentleBoss",
"Korean_ColdGirl",
"Korean_HaughtyLady",
"Korean_CharmingElderSister",
"Korean_IntellectualMan",
"Korean_CaringWoman",
"Korean_WiseTeacher",
"Korean_ConfidentBoss",
"Korean_AthleticGirl",
"Korean_PossessiveMan",
"Korean_GentleWoman",
"Korean_CockyGuy",
"Korean_ThoughtfulWoman",
"Korean_OptimisticYouth",
"Portuguese_AnxiousMan",
"Portuguese_Matureresearcher",
"Portuguese_EnergeticGirl",
"Portuguese_FunnyGuy",
"Portuguese_Nuttylady",
"Portuguese_Deep-tonedMan",
"Portuguese_SentimentalLady",
"Portuguese_BossyLeader",
"Portuguese_Wiselady",
"Portuguese_Strong-WilledBoy",
"Portuguese_Deep-VoicedGentleman",
"Portuguese_UpsetGirl",
"Portuguese_PassionateWarrior",
"Portuguese_AnimeCharacter",
"Portuguese_ConfidentWoman",
"Portuguese_AngryMan",
"Portuguese_CaptivatingStoryteller",
"Portuguese_Godfather",
"Portuguese_ReservedYoungMan",
"Portuguese_SmartYoungGirl",
"Portuguese_Kind-heartedGirl",
"Portuguese_Pompouslady",
"Portuguese_Grinch",
"Portuguese_Debator",
"Portuguese_SweetGirl",
"Portuguese_AttractiveGirl",
"Portuguese_ThoughtfulMan",
"Portuguese_PlayfulGirl",
"Portuguese_GorgeousLady",
"Portuguese_LovelyLady",
"Portuguese_SereneWoman",
"Portuguese_SadTeen",
"Portuguese_MaturePartner",
"Portuguese_Comedian",
"Portuguese_NaughtySchoolgirl",
"Portuguese_Narrator",
"Portuguese_ToughBoss",
"Portuguese_Fussyhostess",
"Portuguese_Dramatist",
"Portuguese_Steadymentor",
"Portuguese_Jovialman",
"Portuguese_CharmingQueen",
"Portuguese_SantaClaus",
"Portuguese_Rudolph",
"Portuguese_Arnold",
"Portuguese_CharmingSanta",
"Portuguese_Ghost",
"Portuguese_HumorousElder",
"Portuguese_CalmLeader",
"Portuguese_GentleTeacher",
"Portuguese_EnergeticBoy",
"Portuguese_ReliableMan",
"Portuguese_SereneElder",
"Portuguese_GrimReaper",
"Portuguese_AssertiveQueen",
"Portuguese_WhimsicalGirl",
"Portuguese_StressedLady",
"Portuguese_FriendlyNeighbor",
"Portuguese_CaringGirlfriend",
"Portuguese_InspiringLady",
"Portuguese_PlayfulSpirit",
"Portuguese_ElegantGirl",
"Portuguese_CompellingGirl",
"Portuguese_PowerfulVeteran",
"Portuguese_SensibleManager",
"Portuguese_ThoughtfulLady",
"Portuguese_TheatricalActor",
"Portuguese_FragileBoy",
"Portuguese_ChattyGirl",
"Portuguese_Conscientiousinstructor",
"Portuguese_RationalMan",
"Portuguese_WiseScholar",
"Portuguese_FrankLady",
"Portuguese_DeterminedManager",
"Portuguese_CharmingLady",
"Russian_HandsomeChildhoodFriend",
"Russian_BrightHeroine",
"Russian_AttractiveGuy",
"Russian_Bad-temperedBoy",
"Spanish_FriendlyNeighbor",
"Spanish_FragileBoy",
"Spanish_UpsetGirl",
"Spanish_Soft-spokenGirl",
"Spanish_CharmingQueen",
"Spanish_Nuttylady",
"Spanish_ElegantGirl",
"Spanish_FascinatingBoy",
"Spanish_FunnyGuy",
"Spanish_PlayfulSpirit",
"Spanish_TheatricalActor",
"Spanish_SereneWoman",
"Spanish_MaturePartner",
"Spanish_CaptivatingStoryteller",
"Spanish_Narrator",
"Spanish_WiseScholar",
"Spanish_Kind-heartedGirl",
"Spanish_DeterminedManager",
"Spanish_BossyLeader",
"Spanish_ReservedYoungMan",
"Spanish_ConfidentWoman",
"Spanish_ThoughtfulMan",
"Spanish_Strong-WilledBoy",
"Spanish_SophisticatedLady",
"Spanish_RationalMan",
"Spanish_AnimeCharacter",
"Spanish_Deep-tonedMan",
"Spanish_Fussyhostess",
"Spanish_SincereTeen",
"Spanish_FrankLady",
"Spanish_Comedian",
"Spanish_Debator",
"Spanish_ToughBoss",
"Spanish_Wiselady",
"Spanish_Steadymentor",
"finnish_male_1_v2",
"hindi_male_1_v2",
"hindi_female_2_v1",
"hindi_female_1_v2",
"Spanish_Jovialman",
"Spanish_SantaClaus",
"Spanish_Rudolph",
"Spanish_Intonategirl",
"Spanish_Arnold",
"Spanish_Ghost",
"Spanish_HumorousElder",
"Spanish_EnergeticBoy",
"Spanish_WhimsicalGirl",
"Spanish_StrictBoss",
"Spanish_ReliableMan",
"Spanish_SereneElder",
"Spanish_AngryMan",
"Spanish_AssertiveQueen",
"Spanish_CaringGirlfriend",
"Spanish_PowerfulSoldier",
"Spanish_PassionateWarrior",
"Spanish_ChattyGirl",
"Spanish_RomanticHusband",
"Spanish_CompellingGirl",
"Spanish_PowerfulVeteran",
"Spanish_SensibleManager",
"Spanish_ThoughtfulLady",
"Turkish_CalmWoman",
"Turkish_Trustworthyman",
"Ukrainian_CalmWoman",
"Ukrainian_WiseScholar",
"Vietnamese_Serene_Man",
"Vietnamese_female_4_v1",
"Vietnamese_male_1_v2",
"Vietnamese_kindhearted_girl",
"Thai_Optimistic_girl",
"Thai_male_1_sample8",
"Thai_Tender_Woman",
"Thai_male_2_sample2",
"Polish_male_1_sample4",
"Polish_male_2_sample3",
"Polish_female_1_sample1",
"Polish_female_2_sample3",
"Romanian_male_1_sample2",
"Romanian_male_2_sample1",
"Romanian_female_1_sample4",
"Romanian_female_2_sample1",
"Greek_female_1_sample1",
"greek_male_1a_v1",
"Greek_female_2_sample3",
"czech_male_1_v1",
"czech_female_5_v7",
"czech_female_2_v2",
"finnish_male_3_v1",
"finnish_female_4_v1",
"Bulgarian_male_2_v1",
"Bulgarian_female_1_v1",
"Danish_male_1_v1",
"Danish_female_1_v1",
"Hebrew_male_1_v1",
"Hebrew_female_1_v1",
"Malay_male_1_v1",
"Malay_female_1_v1",
"Malay_female_2_v1",
"Persian_male_1_v1",
"Persian_female_1_v1",
"Slovak_male_1_v1",
"Slovak_female_1_v1",
"Swedish_male_1_v1",
"Swedish_female_1_v1",
"Croatian_male_1_v1",
"Croatian_female_1_v1",
"Filipino_male_1_v1",
"Filipino_female_1_v1",
"Hungarian_male_1_v1",
"Hungarian_female_1_v1",
"Norwegian_male_1_v1",
"Norwegian_female_1_v1",
"Slovenian_male_1_v1",
"Slovenian_female_1_v2",
"Catalan_male_1_v1",
"Catalan_female_1_v1",
"Nynorsk_male_1_v1",
"Nynorsk_female_1_v1",
"Tamil_male_1_v1",
"Tamil_female_1_v1",
"Afrikaans_male_1_v1",
"Afrikaans_female_1_v1"
],
"description": "Desired voice ID. Use a voice ID you have trained (https://muapi.ai/playground/minimax-voice-clone), or one of the following system voice IDs",
"type": "string",
"typing": true,
"title": "Voice ID",
"name": "voice_id",
"default": "Friendly_Person"
},
"speed": {
"title": "Speed",
"name": "speed",
"type": "int",
"description": "Speech speed. Range: 0.5-2.0, where 1.0 is normal speed.",
"default": 1,
"minValue": 0.5,
"maxValue": 2,
"step": 0.01
},
"volume": {
"title": "Volume",
"name": "volume",
"type": "int",
"description": "Speech volume. Range: 0.1-10.0, where 1.0 is normal volume.",
"default": 1,
"minValue": 0.1,
"maxValue": 10,
"step": 0.01
},
"pitch": {
"title": "Pitch",
"name": "pitch",
"type": "int",
"description": "Speech pitch. Range: -12 to 12, where 0 is normal pitch.",
"default": 0,
"minValue": -12,
"maxValue": 12,
"step": 1
},
"emotion": {
"enum": [
"happy",
"sad",
"angry",
"fearful",
"disgusted",
"surprised",
"neutral"
],
"title": "Emotion",
"name": "emotion",
"type": "string",
"description": "The emotion of the generated speech.",
"default": "surprised"
},
"english_normalization": {
"type": "boolean",
"title": "English Normalization",
"name": "english_normalization",
"description": "This parameter supports English text normalization, which improves performance in number-reading scenarios.",
"default": false
},
"sample_rate": {
"enum": [
8000,
16000,
22050,
24000,
32000,
44100
],
"type": "integer",
"title": "Sample Rate",
"name": "sample_rate",
"description": "Sample rate of generated sound.",
"default": 8000
},
"bitrate": {
"enum": [
32000,
64000,
128000,
256000
],
"type": "integer",
"title": "Bitrate",
"name": "bitrate",
"description": "Bitrate of generated sound.",
"default": 32000
},
"channel": {
"enum": [
1,
2
],
"type": "integer",
"title": "Channel",
"name": "channel",
"description": "he number of channels of the generated audio. 1: mono, 2: stereo.",
"default": 1
},
"format": {
"enum": [
"mp3",
"wav",
"pcm",
"flac"
],
"type": "string",
"title": "Format",
"name": "format",
"description": "Format of generated sound.",
"default": "mp3"
},
"language_boost": {
"enum": [
"Chinese",
"Chinese,Yue",
"English",
"Arabic",
"Russian",
"Spanish",
"French",
"Portuguese",
"German",
"Turkish",
"Dutch",
"Ukrainian",
"Vietnamese",
"Indonesian",
"Japanese",
"Italian",
"Korean",
"Thai",
"Polish",
"Romanian",
"Greek",
"Czech",
"Finnish",
"Hindi",
"Bulgarian",
"Danish",
"Hebrew",
"Malay",
"Persian",
"Slovak",
"Swedish",
"Croatian",
"Filipino",
"Hungarian",
"Norwegian",
"Slovenian",
"Catalan",
"Nynorsk",
"Tamil",
"Afrikaans",
"auto"
],
"title": "Language Boost",
"name": "language_boost",
"type": "string",
"description": "Enhance the ability to recognize specified languages and dialects.",
"default": "auto"
}
},
"title": "BaseInput",
"required": [
"prompt",
"voice_id"
],
"endpoint_url": "minimax-speech-2.6-turbo"
}
}
}
},
{
"name": "gpt-image-1.5",
"category": "Text to Image",
"variant": "Text to Image",
"family": "gpt-1.5",
"group_of": "image",
"description": "GPT-Image-1.5 is a high-quality text-to-image generation model designed for rich visual reasoning, detailed compositions, and strong prompt understanding. It excels at complex scenes, symbolic imagery, cinematic lighting, surreal concepts, product visuals, and imaginative world-building while maintaining coherence and fine detail.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"1:1",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"quality": {
"enum": [
"low",
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated image.",
"default": "medium"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "gpt-image-1.5"
}
}
}
},
{
"name": "wan2.6-image-to-video",
"category": "Image to Video",
"variant": "Image to Video",
"family": "wan2.6",
"group_of": "video",
"description": "WAN 2.6 Image-to-Video converts a single still image into a smooth, cinematic video clip. It preserves the original image\u2019s composition, lighting, and style while adding natural motion, depth parallax, atmospheric effects, and gentle camera movement.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "Audio URL to guide generation (optional).",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"enum": [
5,
10,
15
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
},
"shot_type": {
"enum": [
"single",
"multi"
],
"title": "Shot Type",
"name": "shot_type",
"type": "string",
"description": "The type of shot to generate.",
"default": "single"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "wan2.6-image-to-video"
}
}
}
},
{
"name": "wan2.6-text-to-video",
"category": "Text to Video",
"variant": "Text to Video",
"family": "wan2.6",
"group_of": "video",
"description": "WAN 2.6 Text-to-Video generates smooth, cinematic videos directly from text prompts. It\u2019s designed for strong scene coherence, atmospheric depth, and fluid camera motion, making it ideal for fantasy and sci-fi worlds, surreal concepts, environmental storytelling, and dramatic visual sequences with rich lighting and motion.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"audio_url": {
"description": "Audio URL to guide generation (optional).",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"enum": [
5,
10,
15
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
},
"shot_type": {
"enum": [
"single",
"multi"
],
"title": "Shot Type",
"name": "shot_type",
"type": "string",
"description": "The type of shot to generate.",
"default": "single"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.6-text-to-video"
}
}
}
},
{
"name": "kling-o1-standard-image-to-video",
"category": "Image to Video",
"variant": "Image to Video [Standard]",
"family": "kling-o1",
"group_of": "video",
"description": "Kling O1 Standard Image-to-Video converts a single still image into a short, natural-looking video clip. It preserves the original image\u2019s composition and lighting while adding subtle camera motion, gentle parallax, and light environmental animation. This mode focuses on realism and stability rather than heavy effects, making it ideal for clean cinematic shots, environments, characters, and product visuals.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-o1-standard-image-to-video"
}
}
}
},
{
"name": "kling-o1-standard-reference-to-video",
"category": "Image to Video",
"variant": "Reference to Video [Standard]",
"family": "kling-o1",
"group_of": "video",
"description": "Kling O1 Standard Reference-to-Video generates a smooth, realistic video using one or multiple reference images as visual guidance. It preserves the visual identity, composition, and lighting from the references while adding subtle camera motion, natural parallax, and light environmental animation. This mode prioritizes stability and realism, making it ideal for character shots, environments, product visuals, and calm cinematic scenes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 7
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
5,
10
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-o1-standard-reference-to-video"
}
}
}
},
{
"name": "kling-o1-standard-video-edit",
"category": "Video to Video",
"variant": "Edit Video [Standard]",
"family": "kling-o1",
"group_of": "video",
"description": "Kling O1 Standard Video-to-Video Edit modifies an existing video while preserving its original structure, motion, and realism. It is designed for subtle, stable edits such as object replacement, background changes, lighting adjustments, or small visual tweaks. This mode prioritizes temporal consistency and natural motion, making it.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 4
},
"keep_original_sound": {
"type": "boolean",
"title": "Keep Original Sound",
"name": "keep_original_sound",
"description": "Select whether to keep the video original sound through the parameter.",
"default": true
}
},
"title": "BaseInput",
"required": [
"prompt",
"video_url"
],
"endpoint_url": "kling-o1-standard-video-edit"
}
}
}
},
{
"name": "any-llm",
"category": "Text to Text",
"variant": "Text to Text",
"family": "llm",
"group_of": "text",
"description": "Any LLM is a versatile large language model for text generation, comprehension, and diverse NLP tasks such as chat and summarization. Ready-to-use REST inference API, best performance, no coldstarts, affordable pricing.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the response",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"system_prompt": {
"description": "System prompt to provide context or instructions to the model.",
"type": "string",
"title": "System Prompt",
"name": "system_prompt"
},
"model": {
"enum": [
"anthropic/claude-3.7-sonnet",
"anthropic/claude-3.5-sonnet",
"anthropic/claude-3-haiku",
"google/gemini-2.5-flash",
"google/gemini-2.0-flash-001",
"google/gemini-2.0-flash-lite-001",
"google/gemini-2.5-flash-preview-09-2025",
"google/gemini-2.0-flash-exp:free",
"google/gemini-2.5-pro",
"openai/gpt-4o",
"openai/gpt-4.1",
"openai/gpt-5-chat",
"meta-llama/llama-3.2-90b-vision-instruct",
"meta-llama/llama-4-maverick",
"meta-llama/llama-4-scout"
],
"title": "Model",
"name": "model",
"type": "string",
"description": "Name of the model to use. Premium models are charged at 10x the rate of standard models, they include: deepseek/deepseek-r1, google/gemini-pro-1.5, openai/gpt-4.1, anthropic/claude-3-5-haiku, openai/gpt-4o, anthropic/claude-3.5-sonnet, openai/o3, meta-llama/llama-3.2-90b-vision-instruct, anthropic/claude-3.7-sonnet, openai/gpt-5-chat.",
"default": "google/gemini-2.5-flash"
},
"reasoning": {
"type": "boolean",
"title": "Reasoning",
"name": "reasoning",
"description": "Should reasoning be the part of the final answer.",
"default": false
},
"priority": {
"enum": [
"throughput",
"latency"
],
"title": "Priority",
"name": "priority",
"type": "string",
"description": "Throughput is the default and is recommended for most use cases. Latency is recommended for use cases where low latency is important.",
"default": "throughput"
},
"temperature": {
"title": "Temperature",
"name": "temperature",
"type": "int",
"description": "This setting influences the variety in the model\u2019s responses. Lower values lead to more predictable and typical responses, while higher values encourage more diverse and less common responses. At 0, the model always gives the same response for a given input.",
"default": 1,
"minValue": 0,
"maxValue": 2,
"step": 0.1
},
"max_tokens": {
"title": "Max Tokens",
"name": "max_tokens",
"type": "int",
"description": "This sets the upper limit for the number of tokens the model can generate in response. It won\u2019t produce more than this limit. The maximum value is the context length minus the prompt length.",
"default": null
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "any-llm-models"
}
}
}
},
{
"name": "openrouter-vision",
"category": "Text to Text",
"variant": "Image to Text",
"family": "llm",
"group_of": "text",
"description": "Any LLM is a versatile large language model for text generation, comprehension, and diverse NLP tasks such as chat and summarization. Ready-to-use REST inference API, best performance, no coldstarts, affordable pricing.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the response",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide image urls. Used for image-to-video generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 4
},
"system_prompt": {
"description": "System prompt to provide context or instructions to the model.",
"type": "string",
"title": "System Prompt",
"name": "system_prompt"
},
"model": {
"enum": [
"google/gemini-2.5-flash",
"anthropic/claude-sonnet-4.5",
"openai/gpt-4o",
"qwen/qwen3-vl-235b-a22b-instruct",
"x-ai/grok-4-fast"
],
"title": "Model",
"name": "model",
"type": "string",
"description": "Name of the model to use. Premium models are charged at 10x the rate of standard models, they include: deepseek/deepseek-r1, google/gemini-pro-1.5, openai/gpt-4.1, anthropic/claude-3-5-haiku, openai/gpt-4o, anthropic/claude-3.5-sonnet, openai/o3, meta-llama/llama-3.2-90b-vision-instruct, anthropic/claude-3.7-sonnet, openai/gpt-5-chat.",
"default": "google/gemini-2.5-flash"
},
"reasoning": {
"type": "boolean",
"title": "Reasoning",
"name": "reasoning",
"description": "Should reasoning be the part of the final answer.",
"default": false
},
"temperature": {
"title": "Temperature",
"name": "temperature",
"type": "int",
"description": "This setting influences the variety in the model's responses. Lower values lead to more predictable and typical responses, while higher values encourage more diverse and less common responses. At 0, the model always gives the same response for a given input.",
"default": 1,
"minValue": 0,
"maxValue": 2,
"step": 0.1
},
"max_tokens": {
"title": "Max Tokens",
"name": "max_tokens",
"type": "int",
"description": "This sets the upper limit for the number of tokens the model can generate in response. It won\u2019t produce more than this limit. The maximum value is the context length minus the prompt length.",
"default": null
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "openrouter-vision"
}
}
}
},
{
"name": "audio-passthrough",
"category": "Text to Audio",
"variant": "Text to Audio",
"family": "audio",
"group_of": "audio",
"description": "Audio Passthrough model.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"audio_url": {
"description": "URL of the input audio.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"make_input": {
"type": "boolean",
"title": "Make Input",
"name": "make_input",
"description": "",
"default": true
}
},
"title": "BaseInput",
"required": [
"audio_url"
],
"endpoint_url": ""
}
}
}
},
{
"name": "kling-v2.6-pro-motion-control",
"category": "Video to Video",
"variant": "Pro Motion Control",
"family": "kling-v2.6",
"group_of": "video",
"description": "Kling v2.6 Pro Motion Control allows precise control over camera movement, subject motion, and scene dynamics during video generation. Instead of leaving motion fully implicit, this mode lets you explicitly define how the camera moves (pan, tilt, orbit, dolly, zoom) and how objects or characters behave over time.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Optional prompt for generating video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image. The dimensions should be less than 300px and less than 10MB.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"video_url": {
"description": "URL of the input video. The dimensions should be less than 300px and less than 10MB.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url",
"video_url"
],
"endpoint_url": "kling-v2.6-pro-motion-control"
}
}
}
},
{
"name": "seedance-v1.5-pro-i2v",
"category": "Image to Video",
"variant": "Image to Video",
"family": "seedance-v1.5-pro",
"group_of": "video",
"description": "Seedance v1.5 Pro Image-to-Video converts a single still image into a smooth cinematic video clip. It preserves the original image\u2019s composition, subject identity, and lighting while adding controlled camera motion, natural parallax, and environmental animation. This mode balances visual quality and motion complexity, making it ideal for cinematic scenes, fantasy worlds, sci-fi environments, and storytelling shots.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 4,
"maxValue": 12,
"step": 1
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio",
"default": true
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "seedance-v1.5-pro-i2v"
}
}
}
},
{
"name": "seedance-v1.5-pro-t2v",
"category": "Text to Video",
"variant": "Text to Video",
"family": "seedance-v1.5-pro",
"group_of": "video",
"description": "Seedance v1.5 Pro Text-to-Video generates high-quality cinematic videos directly from text prompts. It focuses on smooth motion, rich atmosphere, and coherent scene structure, making it ideal for fantasy worlds, sci-fi environments, surreal visuals, and cinematic storytelling shots with detailed lighting and depth.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 4,
"maxValue": 12,
"step": 1
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio",
"default": true
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-v1.5-pro-t2v"
}
}
}
},
{
"name": "seedance-v1.5-pro-i2v-fast",
"category": "Image to Video",
"variant": "Image to Video [Fast]",
"family": "seedance-v1.5-pro",
"group_of": "video",
"description": "Seedance v1.5 Pro Image-to-Video Fast converts a single still image into a short cinematic video with quick generation speed. It preserves the original image\u2019s composition, subject identity, and lighting while adding simple camera motion, light parallax, and subtle environmental animation.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 4,
"maxValue": 12,
"step": 1
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio",
"default": true
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "seedance-v1.5-pro-i2v-fast"
}
}
}
},
{
"name": "seedance-v1.5-pro-t2v-fast",
"category": "Text to Video",
"variant": "Text to Video [Fast]",
"family": "seedance-v1.5-pro",
"group_of": "video",
"description": "Seedance v1.5 Pro Text-to-Video Fast generates short cinematic videos directly from text with an emphasis on speed and stability. It produces coherent scenes with simple camera motion, light environmental animation, and consistent lighting.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 4,
"maxValue": 12,
"step": 1
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio",
"default": true
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-v1.5-pro-t2v-fast"
}
}
}
},
{
"name": "seedance-v1.5-pro-video-extend",
"category": "Video to Video",
"variant": "Video Extend",
"family": "seedance-v1.5-pro",
"group_of": "video",
"description": "Seedance v1.5 Pro Video Extend continues an existing video by generating additional frames that match the original scene\u2019s style, lighting, motion, and mood. It is designed for smooth temporal consistency, making it ideal for extending cinematic shots, atmospheric scenes, or slow camera moves without introducing visual jumps or style changes.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video to extend.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"resolution": {
"enum": [
"480p",
"720p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 4,
"maxValue": 12,
"step": 1
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio",
"default": true
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-v1.5-pro-video-extend"
}
}
}
},
{
"name": "seedance-v1.5-pro-video-extend-fast",
"category": "Video to Video",
"variant": "Video Extend [Fast]",
"family": "seedance-v1.5-pro",
"group_of": "video",
"description": "Seedance v1.5 Pro Video Extend Fast quickly extends an existing video by generating a short continuation that matches the original style, motion, and lighting. This mode prioritizes fast output and smooth continuity with minimal new motion, making it ideal for previews, quick edits, and lightweight shot extensions without complex effects.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"video_url": {
"description": "URL of the input video to extend.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
},
"resolution": {
"enum": [
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 4,
"maxValue": 12,
"step": 1
},
"generate_audio": {
"type": "boolean",
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio",
"default": true
},
"camera_fixed": {
"type": "boolean",
"title": "Camera Fixed",
"name": "camera_fixed",
"description": "Whether to fix the camera position",
"default": false
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "seedance-v1.5-pro-video-extend-fast"
}
}
}
},
{
"name": "qwen-image-edit-2511",
"category": "Image to Image",
"variant": "Edit Image 2511",
"family": "qwen",
"group_of": "image",
"description": "Qwen Image Edit 2511 performs precise, instruction-driven edits on an existing image while preserving composition, lighting, and overall style. It\u2019s well-suited for object replacement, material changes, localized edits, and subtle scene adjustments with strong visual consistency and minimal artifacts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "List of URLs of input images for editing.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 3
},
"width": {
"type": "integer",
"title": "Width",
"name": "width",
"description": "Width of the image in pixels",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"type": "integer",
"title": "Height",
"name": "height",
"description": "Height of the image in pixels",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "qwen-image-edit-2511"
}
}
}
},
{
"name": "wan2.6-text-to-image",
"category": "Text to Image",
"variant": "Text to Image",
"family": "wan2.6",
"group_of": "image",
"description": "WAN 2.6 Text-to-Image generates detailed, cinematic still images from text prompts. It focuses on strong composition, atmospheric lighting, and clear subject structure, making it suitable for fantasy and sci-fi environments, surreal concepts, architectural visuals, and dramatic world-building imagery.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"title": "Width",
"name": "width",
"type": "int",
"description": "Width of the output image.",
"default": 1024,
"minValue": 768,
"maxValue": 1440,
"step": 1
},
"height": {
"title": "Height",
"name": "height",
"type": "int",
"description": "Height of the output image.",
"default": 1024,
"minValue": 768,
"maxValue": 1440,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "wan2.6-text-to-image"
}
}
}
},
{
"name": "wan2.6-image-edit",
"category": "Image to Image",
"variant": "Edit Image",
"family": "wan2.6",
"group_of": "image",
"description": "WAN 2.6 Image Edit applies targeted, instruction-based edits to an existing image while preserving composition, perspective, and lighting. It\u2019s ideal for object replacement, material changes, environment tweaks, and style adjustments with clean integration and minimal artifacts\u2014keeping the original scene coherent and cinematic.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 3
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "wan2.6-image-edit"
}
}
}
},
{
"name": "qwen-text-to-image-2512",
"category": "Image to Image",
"variant": "Text to Image 2512",
"family": "qwen",
"group_of": "image",
"description": "Qwen Image Text-to-Image 2512 generates high-resolution, visually consistent images from text prompts. It focuses on strong scene structure, clean composition, and atmospheric lighting, making it well-suited for cinematic environments, surreal concepts, fantasy and sci-fi worlds.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"width": {
"type": "integer",
"title": "Width",
"name": "width",
"description": "Width of the image in pixels",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
},
"height": {
"type": "integer",
"title": "Height",
"name": "height",
"description": "Height of the image in pixels",
"default": 1024,
"minValue": 256,
"maxValue": 1536,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "qwen-text-to-image-2512"
}
}
}
},
{
"name": "gpt-image-1.5-edit",
"category": "Image to Image",
"variant": "Image to Image",
"family": "gpt-1.5",
"group_of": "image",
"description": "GPT-Image-1.5 Edit applies precise, instruction-based modifications to an existing image while preserving composition, lighting, perspective, and visual coherence. It\u2019s well-suited for object replacement, concept evolution, symbolic edits, and creative transformations that feel natural and intentional rather than destructive.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt for edit image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "Upload or provide reference images. Used for image-to-image generation.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 10
},
"aspect_ratio": {
"enum": [
"1:1",
"2:3",
"3:2"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output image.",
"default": "1:1"
},
"quality": {
"enum": [
"low",
"medium",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated image.",
"default": "medium"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "gpt-image-1.5-edit"
}
}
}
},
{
"name": "kling-v2.6-std-motion-control",
"category": "Video to Video",
"variant": "Std Motion Control",
"family": "kling-v2.6",
"group_of": "video",
"description": "Kling v2.6 Pro Motion Control allows precise control over camera movement, subject motion, and scene dynamics during video generation. Instead of leaving motion fully implicit, this mode lets you explicitly define how the camera moves (pan, tilt, orbit, dolly, zoom) and how objects or characters behave over time.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Optional prompt for generating video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image. The dimensions should be less than 300px and less than 10MB.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"video_url": {
"description": "URL of the input video. The dimensions should be less than 300px and less than 10MB.",
"field": "video",
"type": "string",
"title": "Video URL",
"name": "video_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url",
"video_url"
],
"endpoint_url": "kling-v2.6-std-motion-control"
}
}
}
},
{
"name": "grok-imagine-image-to-image",
"category": "Image to Image",
"variant": "Image to Image",
"family": "grok",
"group_of": "image",
"description": "Grok Imagine Image-to-Image transforms an existing image using natural language instructions while preserving scene structure, perspective, and lighting. It is ideal for object replacement, environment evolution, concept re-imagining, and creative edits that feel grounded and visually coherent rather than over-stylized.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image. The dimensions should be less than 300px and less than 10MB.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "grok-imagine-image-to-image"
}
}
}
},
{
"name": "Api Node",
"category": "Image to Image",
"variant": "Image to Image",
"family": "wavespeed",
"group_of": "api",
"description": "",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"model_url": {
"description": "Url of the wavespeed model",
"type": "string",
"format": "text",
"title": "Model URL",
"name": "model_url"
},
"api_key": {
"description": "API key for authentication",
"type": "string",
"format": "text",
"title": "API Key",
"name": "api_key"
},
"params": {}
},
"title": "BaseInput",
"required": [
"model_url",
"api_key"
],
"endpoint_url": ""
}
}
}
},
{
"name": "ltx-2-19b-image-to-video",
"category": "Image to Video",
"variant": "Std Image to Video",
"family": "ltx",
"group_of": "video",
"description": "LTX-2-19B Image-to-Video animates a single image into a coherent cinematic clip with strong temporal stability. It preserves composition and lighting while adding controlled camera motion, realistic parallax, and subtle environmental dynamics\u2014well suited for grounded scenes, near-future concepts, and story beats.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 20,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "ltx-2-19b-image-to-video"
}
}
}
},
{
"name": "ltx-2-19b-text-to-video",
"category": "Text to Video",
"variant": "Std Text to Video",
"family": "ltx",
"group_of": "video",
"description": "LTX-2-19B Text-to-Video generates coherent cinematic videos directly from text, with an emphasis on temporal stability, natural motion, and conceptual clarity. It works best when the scene has a strong visual idea where motion reinforces meaning rather than overwhelming it.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated video",
"default": "16:9"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 5,
"maxValue": 20,
"step": 1
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "ltx-2-19b-text-to-video"
}
}
}
},
{
"name": "veo3.1-4k-video",
"category": "Text to Video",
"variant": "4k Video",
"family": "veo3.1",
"group_of": "video",
"description": "Get the ultra-high-definition 4K version of a Veo3.1 video generation task. This model is optimized for producing crisp, detailed videos suitable for professional and cinematic applications. It enhances visual fidelity while maintaining temporal coherence and realistic motion.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"request_id": {
"description": "Request ID of the original video generation. Must be a valid Id returned from the video generation interface.",
"format": "text",
"type": "string",
"title": "Request Id",
"name": "request_id",
"placeholder": "a8a09145bde3fb496ecd00ce4777a295"
}
},
"title": "BaseInput",
"required": [
"request_id"
],
"endpoint_url": "veo3.1-4k-video"
}
}
}
},
{
"name": "flux-2-klein-4b",
"category": "Text to Image",
"variant": "Text to Image [Klein 4B]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Klein-4B is a lightweight, fast text-to-image model optimized for clear subject rendering, good prompt adherence, and efficient generation. It works best with simple compositions, everyday scenes, and cute or friendly visuals, making it ideal for UI graphics, demos, thumbnails, mascots, and quick creative iterations.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated image",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-2-klein-4b"
}
}
}
},
{
"name": "flux-2-klein-4b-edit",
"category": "Image to Image",
"variant": "Edit Image [Klein 4B]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Klein-4B Edit applies lightweight, instruction-based edits to an existing image. It\u2019s best for clear object swaps, small visual changes, and cute enhancements while preserving the original scene\u2019s layout and lighting. Ideal for fast edits, UI demos, and simple creative tweaks.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "List of URLs of input images for editing.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 4
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated image",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-2-klein-4b-edit"
}
}
}
},
{
"name": "flux-2-klein-9b",
"category": "Text to Image",
"variant": "Text to Image [Klein 9B]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Klein-9B is a mid-size text-to-image model that balances detail quality and generation speed. It handles richer lighting, better textures, and more nuanced scenes than smaller variants, while still working well with clear, grounded prompts. Ideal for polished illustrations, product visuals, mascots, and everyday scenes with character.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated image",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "flux-2-klein-9b"
}
}
}
},
{
"name": "flux-2-klein-9b-edit",
"category": "Image to Image",
"variant": "Edit Image [Klein 9B]",
"family": "flux-2",
"group_of": "image",
"description": "Flux-2-Klein-9B Edit performs higher-quality image edits with better detail retention, lighting consistency, and texture handling compared to smaller variants. It\u2019s well-suited for cute character edits, object additions, and visual refinements that need to look natural and polished while keeping the original scene intact.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image, what you want the final edited image to look like.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"images_list": {
"description": "List of URLs of input images for editing.",
"field": "images_list",
"type": "array",
"items": {
"type": "string"
},
"title": "Image URLs",
"name": "images_list",
"maxItems": 4
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated image",
"default": "1:1"
}
},
"title": "BaseInput",
"required": [
"prompt",
"images_list"
],
"endpoint_url": "flux-2-klein-9b-edit"
}
}
}
},
{
"name": "agentic-architect",
"category": "Text to Text",
"variant": "Architect",
"family": "workflow",
"group_of": "tools",
"description": "Agentic Workflow Architect for generating and editing workflows.",
"input_schema": {}
},
{
"name": "add-image-watermark",
"category": "Image to Image",
"variant": "Add Image Watermark",
"family": "watermark",
"group_of": "image-tools",
"description": "Add custom watermark to images with adjustable position, opacity, and size. Free local processing using PIL.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"image_url": {
"type": "string",
"format": "uri",
"title": "Source Image",
"description": "URL of the image to watermark",
"name": "image_url",
"field": "image"
},
"watermark_image_url": {
"type": "string",
"format": "uri",
"title": "Watermark Image",
"description": "URL of the watermark image (PNG with transparency recommended)",
"name": "watermark_image_url",
"field": "image"
},
"position": {
"enum": [
"top-left",
"top-right",
"bottom-left",
"bottom-right",
"center"
],
"type": "string",
"title": "Position",
"name": "position",
"description": "Position of the watermark on the image",
"default": "bottom-right"
},
"opacity": {
"type": "number",
"minimum": 0,
"maximum": 1,
"title": "Opacity",
"name": "opacity",
"description": "Watermark transparency (0 = invisible, 1 = fully opaque)",
"default": 0.7
},
"scale": {
"type": "number",
"minimum": 0.1,
"maximum": 1,
"title": "Scale",
"name": "scale",
"description": "Watermark size relative to image (0.1 = 10%, 1.0 = 100%)",
"default": 0.2
}
},
"required": [
"image_url",
"watermark_image_url"
],
"endpoint_url": "add-image-watermark"
}
}
}
},
{
"name": "add-video-watermark",
"category": "Video to Video",
"variant": "Add Video Watermark",
"family": "watermark",
"group_of": "video-tools",
"description": "Add custom watermark to videos with adjustable position, opacity, and size. Free local processing using FFmpeg.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"type": "string",
"format": "uri",
"title": "Source Video",
"description": "URL of the video to watermark",
"name": "video_url",
"field": "video"
},
"watermark_image_url": {
"type": "string",
"format": "uri",
"title": "Watermark Image",
"description": "URL of the watermark image (PNG with transparency recommended)",
"name": "watermark_image_url",
"field": "image"
},
"position": {
"enum": [
"top-left",
"top-right",
"bottom-left",
"bottom-right",
"center"
],
"type": "string",
"title": "Position",
"name": "position",
"description": "Position of the watermark on the video",
"default": "bottom-right"
},
"opacity": {
"type": "number",
"minimum": 0,
"maximum": 1,
"title": "Opacity",
"name": "opacity",
"description": "Watermark transparency (0 = invisible, 1 = fully opaque)",
"default": 0.7
},
"scale": {
"type": "number",
"minimum": 0.1,
"maximum": 1,
"title": "Scale",
"name": "scale",
"description": "Watermark size relative to video (0.1 = 10%, 1.0 = 100%)",
"default": 0.15
}
},
"required": [
"video_url",
"watermark_image_url"
],
"endpoint_url": "add-video-watermark"
}
}
}
},
{
"name": "ltx-2-19b-lipsync",
"category": "Audio to Video",
"variant": "Audio to Video",
"family": "ltx",
"group_of": "video",
"description": "LTX-2-19B LipSync generates a realistic talking video by synchronizing a person\u2019s mouth movements to an input audio clip. It preserves facial identity, head position, lighting, and natural expressions while producing accurate lip motion, subtle blinking, and stable temporal consistency. Ideal for avatars, dubbing, dialogue replacement, and character narration.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "The prompt to generate the video",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"audio_url": {
"description": "The URL for uploading audio files.",
"field": "audio",
"type": "string",
"title": "Audio URL",
"name": "audio_url"
},
"resolution": {
"enum": [
"480p",
"720p",
"1080p"
],
"title": "Resolution",
"name": "resolution",
"type": "string",
"description": "The resolution of the generated video.",
"default": "720p"
}
},
"title": "BaseInput",
"required": [
"audio_url"
],
"endpoint_url": "ltx-2-19b-lipsync"
}
}
}
},
{
"name": "z-image-base",
"category": "Text to Image",
"variant": "Text to Image Base",
"family": "z-image",
"group_of": "image",
"description": "Z-Image Base is a general-purpose text-to-image model designed for reliable, high-quality image generation from natural language prompts. It focuses on clear composition, good prompt adherence, and versatile output across everyday scenes, product-style visuals, characters, and creative concepts.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the image.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1",
"3:4",
"4:3",
"21:9",
"9:21"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated image",
"default": "1:1"
},
"strength": {
"title": "Strength",
"name": "strength",
"type": "int",
"description": "Controls the strength of the transformation. Higher values produce outputs more different from the input image.",
"default": 0.6,
"minValue": 0,
"maxValue": 1,
"step": 0.01
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "z-image-base"
}
}
}
},
{
"name": "ai-clipping",
"category": "Video to Video",
"variant": "AI Clipping",
"family": "video",
"group_of": "video",
"description": "Convert long-form videos into engaging short clips using AI clipping.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"video_url": {
"type": "string",
"title": "Video URL",
"field": "video",
"name": "video_url",
"description": "The URL of the video to be clipped."
},
"num_highlights": {
"type": "integer",
"title": "Number of Highlights",
"default": 3,
"name": "num_highlights",
"description": "Number of highlights to extract from the video."
},
"aspect_ratio": {
"type": "string",
"title": "Aspect Ratio",
"enum": [
"9:16",
"1:1",
"4:5"
],
"default": "9:16",
"name": "aspect_ratio",
"description": "The aspect ratio for the resulting clips."
},
"return_coordinates_only": {
"type": "boolean",
"title": "Return Coordinates Only",
"default": false,
"name": "return_coordinates_only",
"description": "Whether to return only the coordinates instead of processed clips."
}
},
"title": "BaseInput",
"required": [
"video_url"
],
"endpoint_url": "ai-clipping"
}
}
}
},
{
"name": "kling-v3.0-pro-image-to-video",
"category": "Image to Video",
"variant": "Image to Video [Pro]",
"family": "kling-v3.0",
"group_of": "video",
"description": "Kling 3.0 Pro Image-to-Video animates a single input image into a high-quality, realistic video with smooth camera motion, natural physics, and strong temporal consistency. It excels at real-world scenes, human motion, environmental details, and cinematic movement while preserving the original image\u2019s structure and lighting.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 15,
"step": 1
},
"generate_audio": {
"type": "boolean",
"default": true,
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio for the video"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v3.0-pro-image-to-video"
}
}
}
},
{
"name": "kling-v3.0-pro-text-to-video",
"category": "Text to Video",
"variant": "Text to Video [Pro]",
"family": "kling-v3.0",
"group_of": "video",
"description": "Kling 3.0 Pro is a high-end video generation model capable of producing longer, smoother, and more realistic cinematic videos with strong motion consistency. It handles complex scenes, realistic physics, natural camera movement, and detailed environments better than earlier versions.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"default": "16:9",
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated video"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 15,
"step": 1
},
"generate_audio": {
"type": "boolean",
"default": true,
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio for the video"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-v3.0-pro-text-to-video"
}
}
}
},
{
"name": "kling-v3.0-standard-image-to-video",
"category": "Image to Video",
"variant": "Image to Video [standard]",
"family": "kling-v3.0",
"group_of": "video",
"description": "Kling 3.0 Standard Image-to-Video animates a single input image into a short, realistic video with smooth, stable motion. It prioritizes temporal consistency, natural physics, and subtle camera movement, making it ideal for everyday scenes, travel moments, people, vehicles, and calm cinematic shots.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"image_url": {
"description": "URL of the input image used to generate video.",
"field": "image",
"type": "string",
"title": "Image URL",
"name": "image_url"
},
"last_image": {
"description": "URL of the input last image.",
"field": "image",
"type": "string",
"title": "Last Image",
"name": "last_image"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 15,
"step": 1
},
"generate_audio": {
"type": "boolean",
"default": true,
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio for the video"
}
},
"title": "BaseInput",
"required": [
"prompt",
"image_url"
],
"endpoint_url": "kling-v3.0-standard-image-to-video"
}
}
}
},
{
"name": "kling-v3.0-standard-text-to-video",
"category": "Text to Video",
"variant": "Text to Video [standard]",
"family": "kling-v3.0",
"group_of": "video",
"description": "Kling 3.0 Standard Text-to-Video generates smooth, realistic videos from text with stable motion and natural behavior. It works best with clear subjects, simple actions, and one continuous scene, making it ideal for cute animals, small actions, and calm cinematic moments.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"description": "Text prompt describing the video.",
"type": "string",
"title": "Prompt",
"name": "prompt"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"default": "16:9",
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "The aspect ratio of the generated video"
},
"duration": {
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5,
"minValue": 3,
"maxValue": 15,
"step": 1
},
"generate_audio": {
"type": "boolean",
"default": true,
"title": "Generate Audio",
"name": "generate_audio",
"description": "Whether to generate audio for the video"
}
},
"title": "BaseInput",
"required": [
"prompt"
],
"endpoint_url": "kling-v3.0-standard-text-to-video"
}
}
}
},
{
"name": "agent-chat",
"category": "Text to Text",
"variant": "Agent Chat",
"family": "infra",
"group_of": "agent-chat",
"description": "Core Agent Chat Interface (Polling Architecture)",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"message": {
"type": "string",
"title": "Message",
"description": "User message to send to the agent"
},
"conversation_id": {
"type": "string",
"title": "Conversation ID",
"description": "Optional conversation ID for multi-turn chat"
}
}
}
}
}
},
{
"name": "seedance-v2.0-t2v",
"category": "Text to Video",
"variant": "Seedance 2.0",
"family": "bytedance",
"group_of": "video",
"description": "Seedance 2.0 is the latest multimodal video generation model by ByteDance, offering advanced camera control, native audio-video sync, and high-resolution output.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"type": "string",
"title": "Prompt",
"description": "The prompt to generate the video"
},
"aspect_ratio": {
"enum": [
"16:9",
"9:16",
"1:1"
],
"title": "Aspect Ratio",
"name": "aspect_ratio",
"type": "string",
"description": "Aspect ratio of the output video.",
"default": "16:9"
},
"duration": {
"enum": [
5,
10,
15
],
"title": "Duration",
"name": "duration",
"type": "int",
"description": "The duration of the generated video in seconds",
"default": 5
},
"quality": {
"enum": [
"basic",
"high"
],
"title": "Quality",
"name": "quality",
"type": "string",
"description": "The quality of the generated video.",
"default": "basic"
}
},
"required": [
"prompt"
],
"endpoint_url": "seedance-v2.0-t2v"
}
}
}
},
{
"name": "seedream-5.0",
"category": "Text to Image",
"isComingSoon": true,
"variant": "Seedream 5.0",
"family": "bytedance",
"group_of": "image",
"description": "Seedream 5.0 is ByteDance\u2019s next-generation text-to-image model featuring visual reasoning, online knowledge integration, native 2K/4K resolution output, and precise prompt understanding with enhanced texture and text rendering.",
"input_schema": {
"schemas": {
"input_data": {
"type": "object",
"properties": {
"prompt": {
"type": "string",
"title": "Prompt",
"description": "Text prompt describing the image to generate"
}
},
"required": [
"prompt"
],
"endpoint_url": "seedream-5.0"
}
}
}
}
]