Merge Video POC ComfyUI nodes

Amp-Thread-ID: https://ampcode.com/threads/T-019fd06e-b484-753d-a063-f3c79085e235
Co-authored-by: Amp <amp@ampcode.com>

# Conflicts:
#	comfy_api_nodes/apis/comfy_cloud.py
#	comfy_api_nodes/nodes_comfy_cloud.py
#	tests-unit/comfy_api_nodes_test/comfy_cloud_test.py
This commit is contained in:
Hunter Senft-Grupp
2026-08-05 05:53:44 +00:00
3 changed files with 245 additions and 0 deletions

View File

@@ -14,6 +14,12 @@ ComfyCloudWorkflow = Literal[
"image.flux-2-reference-edit.v1",
"image.qwen-image-edit-2511.v1",
"image.seedvr2-image-upscale.v1",
"video.minimax-h3-text-sound.v1",
"video.minimax-h3-image-sound.v1",
"video.ltx-2-3-image-audio-performance.v1",
"video.ltx-2-3-first-last-frame.v1",
"video.wan-2-2-14b-first-last-frame.v1",
"video.scail-2-character-replacement.v1",
]
@@ -21,14 +27,24 @@ class ComfyCloudWorkflowInputs(BaseModel):
prompt: str | None = Field(None)
image_url: str | None = Field(None)
assets: dict[str, "ComfyCloudAssetInput"] | None = Field(None)
audio_url: str | None = Field(None)
first_frame_url: str | None = Field(None)
last_frame_url: str | None = Field(None)
reference_character_url: str | None = Field(None)
driving_video_url: str | None = Field(None)
instruction: str | None = Field(None)
prompt_enhance: bool | None = Field(None)
enhance_prompt: bool | None = Field(None)
negative_prompt: str | None = Field(None)
aspect_ratio: str | None = Field(None)
duration_seconds: float | None = Field(None)
guidance: float | None = Field(None)
quality_mode: str | None = Field(None)
seed: int | None = Field(None, ge=0, le=0xFFFFFFFFFFFFFFFF)
scale: str | None = Field(None)
scene_prompt: str | None = Field(None)
driving_subject: str | None = Field(None)
reference_subject: str | None = Field(None)
class ComfyCloudAssetInput(BaseModel):

View File

@@ -18,8 +18,11 @@ from comfy_api_nodes.util import (
get_number_of_images,
poll_op,
sync_op,
upload_audio_to_comfyapi,
upload_image_to_comfyapi,
upload_video_to_comfyapi,
validate_string,
validate_video_frame_count,
)
@@ -399,6 +402,170 @@ class ComfyCloudSeedVR2ImageUpscaleNode(_ComfyCloudWorkflowNode):
)
async def _run_video_workflow(cls: type[IO.ComfyNode], workflow: ComfyCloudWorkflow, inputs: ComfyCloudWorkflowInputs) -> IO.NodeOutput:
task = await sync_op(cls, _GENERATE_ENDPOINT, response_model=ComfyCloudGenerateResponse, data=ComfyCloudGenerateRequest(workflow=workflow, inputs=inputs))
result = await poll_op(
cls,
ApiEndpoint(path=task.polling_url),
response_model=ComfyCloudStatusResponse,
status_extractor=lambda response: response.status,
progress_extractor=lambda response: response.progress,
cancel_endpoint=ApiEndpoint(path=task.cancel_url, method="POST"),
)
if not result.output_url:
detail = f": {result.error}" if result.error else ""
raise RuntimeError(f"Comfy Cloud task {result.task_id} completed without an output URL{detail}")
return IO.NodeOutput(await download_url_to_video_output(result.output_url, cls=cls))
def _video_schema(node_id: str, display_name: str, inputs: list[IO.Input]) -> IO.Schema:
return IO.Schema(
node_id=node_id,
display_name=display_name,
category="partner/video/Comfy Cloud",
inputs=inputs,
outputs=[IO.Video.Output()],
hidden=[IO.Hidden.auth_token_comfy_org, IO.Hidden.api_key_comfy_org, IO.Hidden.unique_id],
is_api_node=True,
)
def _video_seed_input(default: int) -> IO.Int.Input:
return IO.Int.Input("seed", default=default, min=0, max=_UINT64_MAX, control_after_generate=True)
class ComfyCloudMiniMaxH3TextSoundNode(IO.ComfyNode):
@classmethod
def define_schema(cls) -> IO.Schema:
return _video_schema(
"ComfyCloudMiniMaxH3TextSoundNode",
"MiniMax H3 Text + Sound",
[
_prompt_input(),
IO.Combo.Input("aspect_ratio", options=["1:1", "2:3", "3:2", "3:4", "4:3", "9:16", "16:9", "21:9"], default="1:1"),
IO.Float.Input("duration_seconds", default=5, min=5, max=15, step=0.01),
_video_seed_input(168866841893410),
],
)
@classmethod
async def execute(cls, prompt: str, aspect_ratio: str, duration_seconds: float, seed: int) -> IO.NodeOutput:
validate_string(prompt, min_length=1, max_length=4096)
return await _run_video_workflow(cls, "video.minimax-h3-text-sound.v1", ComfyCloudWorkflowInputs(prompt=prompt, aspect_ratio=aspect_ratio, duration_seconds=duration_seconds, seed=seed))
class ComfyCloudMiniMaxH3ImageSoundNode(IO.ComfyNode):
@classmethod
def define_schema(cls) -> IO.Schema:
return _video_schema(
"ComfyCloudMiniMaxH3ImageSoundNode",
"MiniMax H3 Image + Sound",
[
IO.Image.Input("image"),
_prompt_input(),
IO.Combo.Input("aspect_ratio", options=["1:1", "2:3", "3:2", "3:4", "4:3", "9:16", "16:9", "21:9"], default="1:1"),
IO.Float.Input("duration_seconds", default=5, min=5, max=15, step=0.01),
_video_seed_input(168866841893410),
],
)
@classmethod
async def execute(cls, image: Input.Image, prompt: str, aspect_ratio: str, duration_seconds: float, seed: int) -> IO.NodeOutput:
validate_string(prompt, min_length=1, max_length=4096)
if get_number_of_images(image) != 1:
raise ValueError("Exactly one input image is required.")
image_url = await upload_image_to_comfyapi(cls, image)
return await _run_video_workflow(cls, "video.minimax-h3-image-sound.v1", ComfyCloudWorkflowInputs(prompt=prompt, image_url=image_url, aspect_ratio=aspect_ratio, duration_seconds=duration_seconds, seed=seed))
class ComfyCloudLTX23ImageAudioPerformanceNode(IO.ComfyNode):
@classmethod
def define_schema(cls) -> IO.Schema:
return _video_schema(
"ComfyCloudLTX23ImageAudioPerformanceNode",
"LTX-2.3 Image + Audio Performance",
[
IO.Image.Input("image"), IO.Audio.Input("audio"), _prompt_input(),
IO.Boolean.Input("enhance_prompt", default=True),
IO.Float.Input("duration_seconds", default=9, min=1, max=15, step=0.01, tooltip="Must not exceed the input audio duration."),
_video_seed_input(225158785956033),
],
)
@classmethod
async def execute(cls, image: Input.Image, audio: Input.Audio, prompt: str, enhance_prompt: bool, duration_seconds: float, seed: int) -> IO.NodeOutput:
validate_string(prompt, min_length=1, max_length=4096)
if get_number_of_images(image) != 1:
raise ValueError("Exactly one input image is required.")
audio_duration = audio["waveform"].shape[-1] / audio["sample_rate"]
if duration_seconds > audio_duration:
raise ValueError(f"Duration ({duration_seconds:g}s) exceeds input audio duration ({audio_duration:.2f}s).")
image_url = await upload_image_to_comfyapi(cls, image)
audio_url = await upload_audio_to_comfyapi(cls, audio)
return await _run_video_workflow(cls, "video.ltx-2-3-image-audio-performance.v1", ComfyCloudWorkflowInputs(prompt=prompt, image_url=image_url, audio_url=audio_url, enhance_prompt=enhance_prompt, duration_seconds=duration_seconds, seed=seed))
class ComfyCloudLTX23FirstLastFrameNode(IO.ComfyNode):
@classmethod
def define_schema(cls) -> IO.Schema:
return _video_schema(
"ComfyCloudLTX23FirstLastFrameNode",
"LTX-2.3 First & Last Frame",
[IO.Image.Input("first_frame"), IO.Image.Input("last_frame"), _prompt_input(), IO.Int.Input("duration_seconds", default=5, min=2, max=10, step=1, tooltip="25 fps; output frame count is duration × 25 + 1."), _video_seed_input(315253765879496)],
)
@classmethod
async def execute(cls, first_frame: Input.Image, last_frame: Input.Image, prompt: str, duration_seconds: int, seed: int) -> IO.NodeOutput:
validate_string(prompt, min_length=1, max_length=4096)
if get_number_of_images(first_frame) != 1 or get_number_of_images(last_frame) != 1:
raise ValueError("Exactly one first frame and one last frame are required.")
first_url = await upload_image_to_comfyapi(cls, first_frame, wait_label="Uploading first frame")
last_url = await upload_image_to_comfyapi(cls, last_frame, wait_label="Uploading last frame")
return await _run_video_workflow(cls, "video.ltx-2-3-first-last-frame.v1", ComfyCloudWorkflowInputs(prompt=prompt, first_frame_url=first_url, last_frame_url=last_url, duration_seconds=duration_seconds, seed=seed))
class ComfyCloudWan22FirstLastFrameNode(IO.ComfyNode):
@classmethod
def define_schema(cls) -> IO.Schema:
return _video_schema(
"ComfyCloudWan22FirstLastFrameNode",
"Wan 2.2 14B First & Last Frame",
[IO.Image.Input("first_frame"), IO.Image.Input("last_frame"), _prompt_input(), IO.String.Input("negative_prompt", multiline=True, default="graph tested Chinese quality negative"), IO.Int.Input("duration_seconds", default=5, min=2, max=8, step=1, tooltip="Graph frame count is floor(duration × 16 + 1)."), _video_seed_input(984937593540091)],
)
@classmethod
async def execute(cls, first_frame: Input.Image, last_frame: Input.Image, prompt: str, negative_prompt: str, duration_seconds: int, seed: int) -> IO.NodeOutput:
validate_string(prompt, min_length=1, max_length=4096)
validate_string(negative_prompt, min_length=0, max_length=2048)
if get_number_of_images(first_frame) != 1 or get_number_of_images(last_frame) != 1:
raise ValueError("Exactly one first frame and one last frame are required.")
first_url = await upload_image_to_comfyapi(cls, first_frame, wait_label="Uploading first frame")
last_url = await upload_image_to_comfyapi(cls, last_frame, wait_label="Uploading last frame")
return await _run_video_workflow(cls, "video.wan-2-2-14b-first-last-frame.v1", ComfyCloudWorkflowInputs(prompt=prompt, negative_prompt=negative_prompt, first_frame_url=first_url, last_frame_url=last_url, duration_seconds=duration_seconds, seed=seed))
class ComfyCloudSCAIL2CharacterReplacementNode(IO.ComfyNode):
@classmethod
def define_schema(cls) -> IO.Schema:
return _video_schema(
"ComfyCloudSCAIL2CharacterReplacementNode",
"SCAIL-2 Character Replacement",
[IO.Image.Input("reference_character"), IO.Video.Input("driving_video", tooltip="Must contain 81157 decoded frames."), _prompt_input("scene_prompt"), IO.String.Input("driving_subject", default=""), IO.String.Input("reference_subject", default="human"), _video_seed_input(1)],
)
@classmethod
async def execute(cls, reference_character: Input.Image, driving_video: Input.Video, scene_prompt: str, driving_subject: str, reference_subject: str, seed: int) -> IO.NodeOutput:
validate_string(scene_prompt, min_length=1, max_length=4096)
validate_string(driving_subject, min_length=1, max_length=256)
validate_string(reference_subject, min_length=1, max_length=256)
if get_number_of_images(reference_character) != 1:
raise ValueError("Exactly one reference character image is required.")
validate_video_frame_count(driving_video, min_frame_count=81, max_frame_count=157)
image_url = await upload_image_to_comfyapi(cls, reference_character)
video_url = await upload_video_to_comfyapi(cls, driving_video)
return await _run_video_workflow(cls, "video.scail-2-character-replacement.v1", ComfyCloudWorkflowInputs(scene_prompt=scene_prompt, driving_subject=driving_subject, reference_subject=reference_subject, reference_character_url=image_url, driving_video_url=video_url, seed=seed))
class ComfyCloudExtension(ComfyExtension):
@override
async def get_node_list(self) -> list[type[IO.ComfyNode]]:
@@ -413,6 +580,12 @@ class ComfyCloudExtension(ComfyExtension):
ComfyCloudFlux2ReferenceEditNode,
ComfyCloudQwenImageEdit2511Node,
ComfyCloudSeedVR2ImageUpscaleNode,
ComfyCloudMiniMaxH3TextSoundNode,
ComfyCloudMiniMaxH3ImageSoundNode,
ComfyCloudLTX23ImageAudioPerformanceNode,
ComfyCloudLTX23FirstLastFrameNode,
ComfyCloudWan22FirstLastFrameNode,
ComfyCloudSCAIL2CharacterReplacementNode,
]

View File

@@ -246,6 +246,62 @@ def test_image_poc_api_declarations_and_extension_registration():
assert {node for node, _, _, _ in IMAGE_POC_NODES} <= registered
@pytest.mark.parametrize(
("node", "input_names"),
[
(nodes_comfy_cloud.ComfyCloudMiniMaxH3TextSoundNode, ["prompt", "aspect_ratio", "duration_seconds", "seed"]),
(nodes_comfy_cloud.ComfyCloudMiniMaxH3ImageSoundNode, ["image", "prompt", "aspect_ratio", "duration_seconds", "seed"]),
(nodes_comfy_cloud.ComfyCloudLTX23ImageAudioPerformanceNode, ["image", "audio", "prompt", "enhance_prompt", "duration_seconds", "seed"]),
(nodes_comfy_cloud.ComfyCloudLTX23FirstLastFrameNode, ["first_frame", "last_frame", "prompt", "duration_seconds", "seed"]),
(nodes_comfy_cloud.ComfyCloudWan22FirstLastFrameNode, ["first_frame", "last_frame", "prompt", "negative_prompt", "duration_seconds", "seed"]),
(nodes_comfy_cloud.ComfyCloudSCAIL2CharacterReplacementNode, ["reference_character", "driving_video", "scene_prompt", "driving_subject", "reference_subject", "seed"]),
],
)
def test_video_node_schemas_expose_only_manifest_inputs(node, input_names):
schema = node.define_schema()
assert schema.is_api_node
assert [input.id for input in schema.inputs] == input_names
assert len(schema.outputs) == 1
assert schema.outputs[0].get_io_type() == "VIDEO"
def test_ltx_performance_stages_image_and_audio(monkeypatch):
run = AsyncMock(return_value=("video-output",))
image_upload = AsyncMock(return_value="https://example.com/image.png")
audio_upload = AsyncMock(return_value="https://example.com/audio.mp4")
monkeypatch.setattr(nodes_comfy_cloud, "_run_video_workflow", run)
monkeypatch.setattr(nodes_comfy_cloud, "upload_image_to_comfyapi", image_upload)
monkeypatch.setattr(nodes_comfy_cloud, "upload_audio_to_comfyapi", audio_upload)
monkeypatch.setattr(nodes_comfy_cloud, "get_number_of_images", lambda image: 1)
audio = {"waveform": torch.zeros(1, 1, 480000), "sample_rate": 48000}
asyncio.run(nodes_comfy_cloud.ComfyCloudLTX23ImageAudioPerformanceNode.execute(object(), audio, "sing", True, 9, 7))
inputs = run.call_args.args[2]
assert inputs.image_url == "https://example.com/image.png"
assert inputs.audio_url == "https://example.com/audio.mp4"
assert inputs.duration_seconds == 9
def test_scail_stages_reference_image_and_driving_video(monkeypatch):
run = AsyncMock(return_value=("video-output",))
image_upload = AsyncMock(return_value="https://example.com/character.png")
video_upload = AsyncMock(return_value="https://example.com/driving.mp4")
video = Mock()
video.get_frame_count.return_value = 100
monkeypatch.setattr(nodes_comfy_cloud, "_run_video_workflow", run)
monkeypatch.setattr(nodes_comfy_cloud, "upload_image_to_comfyapi", image_upload)
monkeypatch.setattr(nodes_comfy_cloud, "upload_video_to_comfyapi", video_upload)
monkeypatch.setattr(nodes_comfy_cloud, "get_number_of_images", lambda image: 1)
asyncio.run(nodes_comfy_cloud.ComfyCloudSCAIL2CharacterReplacementNode.execute(object(), video, "park", "woman", "human", 1))
inputs = run.call_args.args[2]
assert inputs.reference_character_url == "https://example.com/character.png"
assert inputs.driving_video_url == "https://example.com/driving.mp4"
video.get_frame_count.assert_called_once()
def test_download_cloud_audio_url_to_audio_input(monkeypatch):
node = nodes_comfy_cloud.ComfyCloudTextToImageNode
downloaded = b"encoded audio"