diff --git a/comfy/ldm/qwen_image21/model.py b/comfy/ldm/qwen_image21/model.py index 8e2d6ad5ab4..b804b9ecad9 100644 --- a/comfy/ldm/qwen_image21/model.py +++ b/comfy/ldm/qwen_image21/model.py @@ -395,6 +395,9 @@ def block_wrap(args): hidden_states = block(hidden_states, mod, pe, attn_fn, prefix_len, transformer_options) for p in patches.get("single_block", []): hidden_states = p({"img": hidden_states, "x": x, "block_index": i, "transformer_options": transformer_options})["img"] + if cache is not None: + # Release dequantized K/V before leaving the block's allocation scope. + del attn_fn, prefix_k, prefix_v comfy.model_prefetch.prefetch_queue_pop(prefetch_queue, x.device, None, malloc_scope="block") comfy.model_prefetch.malloc_graph_end() diff --git a/comfy_api_nodes/apis/anthropic.py b/comfy_api_nodes/apis/anthropic.py index 5e74d23558a..0ef98f18fbf 100644 --- a/comfy_api_nodes/apis/anthropic.py +++ b/comfy_api_nodes/apis/anthropic.py @@ -36,7 +36,7 @@ class AnthropicMessage(BaseModel): class AnthropicThinkingConfig(BaseModel): - type: Literal["enabled", "disabled", "adaptive"] = Field(...) + type: Literal["enabled", "disabled", "adaptive", "between_tools"] = Field(...) budget_tokens: int | None = Field( None, ge=1024, description="Reasoning budget in tokens. Used when type is 'enabled'. Must be less than max_tokens.", diff --git a/comfy_api_nodes/apis/bfl.py b/comfy_api_nodes/apis/bfl.py index abc5d043de7..a4a6708e90a 100644 --- a/comfy_api_nodes/apis/bfl.py +++ b/comfy_api_nodes/apis/bfl.py @@ -184,3 +184,14 @@ class BFLFluxVideoEditRequest(BaseModel): video: str = Field(..., description="MP4 (URL or base64), at most 15 seconds and 50 MiB.") prompt: str = Field(..., description="Edit instruction, 1 to 4096 characters once trimmed.") safety_tolerance: int = Field(4, ge=0, le=4) + + +class Flux3ImageRequest(BaseModel): + model_config = ConfigDict(extra="forbid") + + prompt: str = Field(...) + images: list[str] | None = Field(None, description="1 to 10 reference images (URL or base64).") + aspect_ratio: str = Field("auto") + resolution: str = Field("1k") + grounding: bool = Field(True, description="Web and image search before generating.") + safety_tolerance: int = Field(2, ge=0, le=4) diff --git a/comfy_api_nodes/apis/ideogram.py b/comfy_api_nodes/apis/ideogram.py index 12c9b2fc398..a0f05a8b57f 100644 --- a/comfy_api_nodes/apis/ideogram.py +++ b/comfy_api_nodes/apis/ideogram.py @@ -264,3 +264,11 @@ class IdeogramV4Request(BaseModel): resolution: str | None = Field(None, description="Output resolution in WIDTHxHEIGHT (e.g. '2048x2048').") rendering_speed: str | None = Field(None, description="Rendering speed: 'TURBO', 'DEFAULT', or 'QUALITY'.") enable_copyright_detection: bool | None = Field(None, description="Opt into post-generation copyright detection.") + + +class Ideogram45Request(BaseModel): + prompt: str + quality: str + seed: int = Field(..., ge=0, le=2147483647) + size: str | None = None + magic_prompt: str | None = None diff --git a/comfy_api_nodes/nodes_anthropic.py b/comfy_api_nodes/nodes_anthropic.py index e627920c599..6be23ec77bb 100644 --- a/comfy_api_nodes/nodes_anthropic.py +++ b/comfy_api_nodes/nodes_anthropic.py @@ -33,6 +33,7 @@ "Opus 4.8": "claude-opus-4-8", "Fable 5.1": "claude-fable-5-1", "Fable 5": "claude-fable-5", + "Sonnet 5.5": "claude-sonnet-5-5", "Sonnet 5": "claude-sonnet-5", "Opus 4.7": "claude-opus-4-7", "Opus 4.6": "claude-opus-4-6", @@ -44,12 +45,13 @@ _THINKING_UNSUPPORTED = {"Haiku 4.5"} # Models that use the newer "adaptive" thinking mode (Opus 4.7+ require it; older models keep the explicit budget API). # Anthropic decides the actual budget when adaptive is used, based on the `output_config.effort` hint. -_ADAPTIVE_THINKING_MODELS = {"Opus 4.8", "Sonnet 5", "Opus 4.7", "Opus 4.6", "Sonnet 4.6"} +_ADAPTIVE_THINKING_MODELS = {"Opus 4.8", "Sonnet 5.5", "Sonnet 5", "Opus 4.7", "Opus 4.6", "Sonnet 4.6"} _ALWAYS_THINKING_MODELS = {"Opus 5.5", "Opus 5", "Fable 5.1", "Fable 5"} -_XHIGH_EFFORT_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5", "Opus 4.7"} +_XHIGH_EFFORT_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5.5", "Sonnet 5", "Opus 4.7"} _MAX_EFFORT_MODELS = _XHIGH_EFFORT_MODELS | {"Opus 4.6", "Sonnet 4.6"} -_EXPLICIT_THINKING_OFF_MODELS = {"Sonnet 5"} -_NO_TEMPERATURE_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5"} +_EXPLICIT_THINKING_OFF_MODELS = {"Sonnet 5.5": "between_tools", "Sonnet 5": "disabled"} +_NO_TEMPERATURE_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5.5", "Sonnet 5"} +_LOW_MAX_TOKENS_MODELS = {"Opus 5.5", "Sonnet 5.5"} # Budget mode (Sonnet 4.5): effort -> reasoning budget in tokens. Must be < max_tokens. # Sized so even the "high" budget fits comfortably under the default max_tokens=32768. @@ -77,7 +79,7 @@ def _claude_model_inputs(model_label: str): IO.Int.Input( "max_tokens", default=32768, - min=4096, + min=1024 if model_label in _LOW_MAX_TOKENS_MODELS else 4096, max=64000, tooltip="Maximum number of tokens to generate (includes reasoning tokens when enabled).", advanced=True, @@ -297,7 +299,7 @@ async def execute( budget = min(budget, max(1024, max_tokens - 1024)) thinking_cfg = AnthropicThinkingConfig(type="enabled", budget_tokens=budget) elif model_label in _EXPLICIT_THINKING_OFF_MODELS: - thinking_cfg = AnthropicThinkingConfig(type="disabled") + thinking_cfg = AnthropicThinkingConfig(type=_EXPLICIT_THINKING_OFF_MODELS[model_label]) image_tensors: list[Input.Image] = [t for t in (images or {}).values() if t is not None] if sum(get_number_of_images(t) for t in image_tensors) > CLAUDE_MAX_IMAGES: diff --git a/comfy_api_nodes/nodes_bfl.py b/comfy_api_nodes/nodes_bfl.py index 33c0be568f6..3b11d294c2a 100644 --- a/comfy_api_nodes/nodes_bfl.py +++ b/comfy_api_nodes/nodes_bfl.py @@ -1,3 +1,4 @@ +import json import math import torch @@ -18,6 +19,7 @@ BFLFluxVTORequest, BFLStatus, Flux2ProGenerateRequest, + Flux3ImageRequest, Flux3ImageToVideoRequest, Flux3TextToVideoRequest, Flux3VideoContinuationRequest, @@ -1616,6 +1618,176 @@ async def execute( return await _bfl_video_execute(cls, _FLUX_VIDEO_EDIT_ENDPOINT, request, poll_via_proxy=True) +_FLUX3_IMAGE_ENDPOINT = ApiEndpoint(path="/proxy/bfl/v1/flux-3-image", method="POST") +_FLUX3_IMAGE_ASPECT_RATIOS = [ + "auto", "21:9", "2:1", "16:9", "3:2", "7:5", "4:3", "5:4", "1:1", "4:5", "3:4", "5:7", "2:3", "9:16", "1:2", "9:21" +] +_FLUX3_IMAGE_RESOLUTIONS = {"0.75K": "768sq", "1K": "1k", "1.5K": "1.5k", "2K": "2k", "4K": "4k"} +_FLUX3_IMAGE_MAX_PROMPT_LENGTH = 15000 + + +def _flux3_box_rows(elements: list) -> list[dict]: + rows = [] + for index, element in enumerate(elements, start=1): + parts = [] + if element.get("type") == "text": + parts.append(f'the text "{element.get("text", "")}"') + if element.get("desc"): + parts.append(element["desc"]) + if element.get("color_palette"): + parts.append("colors " + ", ".join(element["color_palette"])) + row = {"id": f"box_{index}", "bbox": element["bbox"]} + if parts: + row["desc"] = ", ".join(parts) + rows.append(row) + return rows + + +class Flux3ImageNode(IO.ComfyNode): + + @classmethod + def define_schema(cls) -> IO.Schema: + return IO.Schema( + node_id="Flux3ImageNode", + display_name="Flux 3 Image", + category="partner/image/BFL", + description="Generates an image with FLUX 3 from a prompt, or edits and combines up to 10 " + "reference images. Refer to the references in the prompt as image 1, image 2, and so on.", + inputs=[ + IO.String.Input( + "prompt", + multiline=True, + default="", + tooltip="What to generate, or the edit to make. The prompt is interpreted and " + "expanded before generation.", + ), + IO.Autogrow.Input( + "images", + template=IO.Autogrow.TemplateNames( + IO.Image.Input("image"), + names=[f"image_{i}" for i in range(1, _FLUX3_MAX_IMAGES + 1)], + min=0, + ), + tooltip="Optional reference images, up to 10 in total, at least 256x256 pixels each.", + ), + IO.Array.Input( + "bounding_boxes", + optional=True, + tooltip="Optional boxes from Create Bounding Boxes that place objects or text in the " + "output. Positions are relative to the canvas, so give it the output's aspect ratio.", + ), + IO.Combo.Input( + "aspect_ratio", + options=_FLUX3_IMAGE_ASPECT_RATIOS, + default="auto", + tooltip="'auto' follows the first reference image, or picks a ratio from the prompt.", + ), + IO.Combo.Input( + "resolution", + options=list(_FLUX3_IMAGE_RESOLUTIONS), + default="2K", + tooltip="Output size at the chosen aspect ratio: 0.75K is about 0.6 megapixels, " + "1K 1 MP, 1.5K 2.4 MP, 2K 4.2 MP, 4K 16.8 MP.", + ), + IO.Boolean.Input( + "grounding", + default=True, + tooltip="Let the model research the prompt with web and image search before generating.", + ), + IO.Int.Input( + "safety_tolerance", + default=4, + min=0, + max=4, + advanced=True, + tooltip="Moderation tolerance, 0 is the strictest.", + ), + IO.Int.Input( + "seed", + default=42, + min=0, + max=0xFFFFFFFF, + control_after_generate=True, + tooltip="Seed to determine if node should re-run; FLUX 3 picks its own seed, so " + "actual results are nondeterministic regardless of this value.", + ), + ], + outputs=[IO.Image.Output()], + hidden=[ + IO.Hidden.auth_token_comfy_org, + IO.Hidden.api_key_comfy_org, + IO.Hidden.unique_id, + ], + is_api_node=True, + price_badge=IO.PriceBadge( + depends_on=IO.PriceBadgeDepends(widgets=["resolution"]), + expr=""" + ( + $prices := {"0.75k": 0.05863, "1k": 0.06864, "1.5k": 0.1001, "2k": 0.143, "4k": 0.86801}; + {"type": "usd", "usd": $lookup($prices, widgets.resolution)} + ) + """, + ), + ) + + @classmethod + async def execute( + cls, + prompt: str, + images: IO.Autogrow.Type, + aspect_ratio: str, + resolution: str, + grounding: bool, + safety_tolerance: int, + seed: int, + bounding_boxes: list | None = None, + ) -> IO.NodeOutput: + validate_string(prompt, field_name="prompt", min_length=1, max_length=_FLUX3_IMAGE_MAX_PROMPT_LENGTH) + if bounding_boxes: + prompt = f"{prompt} {json.dumps(_flux3_box_rows(bounding_boxes), ensure_ascii=False)}" + if len(prompt) > _FLUX3_IMAGE_MAX_PROMPT_LENGTH: + raise ValueError( + f"The prompt together with the bounding boxes is {len(prompt)} characters long, " + f"the limit is {_FLUX3_IMAGE_MAX_PROMPT_LENGTH}. Shorten the prompt or the box descriptions." + ) + reference_images = _flux3_collect_images(images, "reference images") + image_urls = None + if reference_images: + image_urls = await upload_images_to_comfyapi( + cls, reference_images, max_images=_FLUX3_MAX_IMAGES, wait_label="Uploading references" + ) + initial_response = await sync_op( + cls, + _FLUX3_IMAGE_ENDPOINT, + response_model=BFLFluxProGenerateResponse, + data=Flux3ImageRequest( + prompt=prompt, + images=image_urls, + aspect_ratio=aspect_ratio, + resolution=_FLUX3_IMAGE_RESOLUTIONS[resolution], + grounding=grounding, + safety_tolerance=safety_tolerance, + ), + ) + response = await poll_op( + cls, + ApiEndpoint(path=_BFL_POLL_PROXY_PATH, query_params={"polling_url": initial_response.polling_url}), + response_model=BFLFluxStatusResponse, + status_extractor=lambda r: r.status, + progress_extractor=lambda r: r.progress, + completed_statuses=[BFLStatus.ready], + failed_statuses=[ + BFLStatus.request_moderated, + BFLStatus.content_moderated, + BFLStatus.error, + BFLStatus.task_not_found, + ], + queued_statuses=[BFLStatus.pending], + max_retries_per_poll=3, + ) + return IO.NodeOutput(await download_url_to_image_tensor(response.result["sample"])) + + class BFLExtension(ComfyExtension): @override async def get_node_list(self) -> list[type[IO.ComfyNode]]: @@ -1635,6 +1807,7 @@ async def get_node_list(self) -> list[type[IO.ComfyNode]]: Flux3VideoContinuationNode, FluxVideoUpscaleNode, FluxVideoEditNode, + Flux3ImageNode, ] diff --git a/comfy_api_nodes/nodes_grok.py b/comfy_api_nodes/nodes_grok.py index c58fb1a29f1..a5e7681048f 100644 --- a/comfy_api_nodes/nodes_grok.py +++ b/comfy_api_nodes/nodes_grok.py @@ -624,24 +624,26 @@ def define_schema(cls): inputs=[ IO.Combo.Input( "model", - options=["grok-imagine-video", "grok-imagine-video-1.5"], + options=["grok-imagine-video", "grok-imagine-video-1.5", "grok-imagine-video-1.5-lite"], + default="grok-imagine-video-1.5-lite", tooltip="The model to use for video generation.", ), IO.String.Input( "prompt", multiline=True, tooltip="Text description of the desired video. " - "Optional for grok-imagine-video-1.5 when an input image is provided.", + "Optional for the grok-imagine-video-1.5 models when an input image is provided.", ), IO.Combo.Input( "resolution", options=["480p", "720p", "1080p"], - tooltip="The resolution of the output video. 1080p is only available for grok-imagine-video-1.5.", + tooltip="The resolution of the output video. 1080p is not available for grok-imagine-video.", ), IO.Combo.Input( "aspect_ratio", options=["auto", "16:9", "4:3", "3:2", "1:1", "2:3", "3:4", "9:16"], - tooltip="The aspect ratio of the output video.", + tooltip="The aspect ratio of the output video. " + "Ignored when an input image is provided; the video follows the image's aspect ratio.", ), IO.Int.Input( "duration", @@ -682,10 +684,13 @@ def define_schema(cls): depends_on=IO.PriceBadgeDepends(widgets=["model", "duration", "resolution"], inputs=["image"]), expr=""" ( + $isLite := widgets.model = "grok-imagine-video-1.5-lite"; $is15 := $contains(widgets.model, "1.5"); - $rate := $is15 - ? (widgets.resolution = "1080p" ? 0.25 : (widgets.resolution = "720p" ? 0.14 : 0.08)) - : (widgets.resolution = "720p" ? 0.07 : 0.05); + $rate := $isLite + ? (widgets.resolution = "1080p" ? 0.14 : (widgets.resolution = "720p" ? 0.03 : 0.02)) + : ($is15 + ? (widgets.resolution = "1080p" ? 0.25 : (widgets.resolution = "720p" ? 0.14 : 0.08)) + : (widgets.resolution = "720p" ? 0.07 : 0.05)); $imgCost := $is15 ? 0.01 : 0.002; $base := $rate * widgets.duration; $total := inputs.image.connected ? $base + $imgCost : $base; @@ -706,14 +711,14 @@ async def execute( seed: int, image: Input.Image | None = None, ) -> IO.NodeOutput: - if resolution == "1080p" and model != "grok-imagine-video-1.5": - raise ValueError(f"1080p resolution is only available for grok-imagine-video-1.5, not '{model}'.") + if resolution == "1080p" and model == "grok-imagine-video": + raise ValueError("1080p resolution is not available for grok-imagine-video.") image_url = None if image is not None: if get_number_of_images(image) != 1: raise ValueError("Only one input image is supported.") image_url = InputUrlObject(url=f"data:image/png;base64,{tensor_to_base64_string(image)}") - if image is None or model != "grok-imagine-video-1.5": + if image is None or model == "grok-imagine-video": validate_string(prompt, strip_whitespace=True, min_length=1) initial_response = await sync_op( cls, diff --git a/comfy_api_nodes/nodes_heygen.py b/comfy_api_nodes/nodes_heygen.py index b1375d14315..28b7c26d916 100644 --- a/comfy_api_nodes/nodes_heygen.py +++ b/comfy_api_nodes/nodes_heygen.py @@ -1,3 +1,5 @@ +import re + import torch from typing_extensions import override @@ -25,11 +27,13 @@ upload_image_to_comfyapi, upload_images_to_comfyapi, upload_video_to_comfyapi, + validate_image_aspect_ratio, validate_string, ) from server import PromptServer _VIDEOS_PATH = "/proxy/heygen/v3/videos" +_MODEL_VIDEOS_PATH = "/proxy/heygen/v3/models/videos" _TRANSLATIONS_PATH = "/proxy/heygen/v3/video-translations" _SPEECH_PATH = "/proxy/heygen/v3/voices/speech" _AVATARS_PATH = "/proxy/heygen/v3/avatars" @@ -42,6 +46,27 @@ for e in ("avatar_iv", "avatar_iii", "avatar_v") } +_REFERENCE_TAG_RE = re.compile(r"(? str: + def repl(match: re.Match) -> str: + kind = match.group(1).lower() + idx = int(match.group(2) or 1) + if not 1 <= idx <= counts[kind]: + raise ValueError( + f"The prompt references @{kind.capitalize()}{idx}, " + f"but only {counts[kind]} reference {kind} inputs are connected." + ) + return f"<{_REFERENCE_LABELS[kind]} {idx}>" + + prev = None + while prev != prompt: + prev = prompt + prompt = _REFERENCE_TAG_RE.sub(repl, prompt) + return prompt + async def _apply_speech_source(cls: type[IO.ComfyNode], payload: dict, speech: dict, require_voice: bool) -> None: """Fill script/audio speech fields of a /v3/videos payload from the DynamicCombo dict.""" @@ -64,11 +89,11 @@ async def _apply_speech_source(cls: type[IO.ComfyNode], payload: dict, speech: d payload["voice_settings"] = {"speed": round(speed, 2)} -async def _create_and_poll_video(cls: type[IO.ComfyNode], payload: dict) -> dict: - """POST a /v3/videos payload, poll until terminal, and return the final video data.""" +async def _create_and_poll_video(cls: type[IO.ComfyNode], create_path: str, payload: dict) -> dict: + """POST a video payload, poll /v3/videos until terminal, and return the final video data.""" created = await sync_op_raw( cls, - ApiEndpoint(path=_VIDEOS_PATH, method="POST"), + ApiEndpoint(path=create_path, method="POST"), data=payload, ) video_id = (created.get("data") or {}).get("video_id") @@ -248,7 +273,7 @@ async def execute( "title": "ComfyUI Talking Photo", } await _apply_speech_source(cls, payload, speech, require_voice=True) - video = await _create_and_poll_video(cls, payload) + video = await _create_and_poll_video(cls, _VIDEOS_PATH, payload) return IO.NodeOutput(await download_url_to_video_output(video["video_url"])) @@ -449,7 +474,7 @@ async def execute( raise ValueError("background_color must be a hex color code like '#00ff00'.") payload["background"] = {"type": "color", "value": background_color} await _apply_speech_source(cls, payload, speech, require_voice=False) - video = await _create_and_poll_video(cls, payload) + video = await _create_and_poll_video(cls, _VIDEOS_PATH, payload) return IO.NodeOutput(await download_url_to_video_output(video["video_url"])) @@ -783,6 +808,272 @@ async def execute( return IO.NodeOutput(audio_bytes_to_audio_input(audio_bytes.getvalue())) +class HeyGenReferenceToVideoNode(IO.ComfyNode): + + @classmethod + def define_schema(cls) -> IO.Schema: + return IO.Schema( + node_id="HeyGenReferenceToVideoNode", + display_name="HeyGen Video 1.0 Reference to Video", + category="partner/video/HeyGen", + description="Generate a video with synchronized dialogue and sound from a text prompt using " + "HeyGen Video 1.0. Optionally connect up to 12 reference images, videos, and audio clips and " + "mention them in the prompt as @Image1, @Video1, @Audio1.", + inputs=[ + IO.DynamicCombo.Input( + "model", + options=[ + IO.DynamicCombo.Option( + "heygen-video-1", + [ + IO.String.Input( + "prompt", + multiline=True, + default="", + tooltip="Description of the video, including any dialogue. Refer to " + "connected references as @Image1, @Video1, @Audio1, numbered per type " + "in input order.", + ), + IO.Int.Input( + "duration", + default=5, + min=5, + max=15, + step=1, + display_mode=IO.NumberDisplay.slider, + tooltip="Duration of the output video in seconds.", + ), + IO.Combo.Input( + "resolution", + options=["768p", "480p"], + tooltip="Output resolution.", + ), + IO.Combo.Input( + "aspect_ratio", + options=["auto", "16:9", "9:16", "1:1", "4:3", "3:4", "21:9"], + tooltip="Output aspect ratio. 'auto' is 16:9 without references and " + "otherwise follows the first reference image (or the first reference " + "video when no images are connected).", + ), + IO.Int.Input( + "seed", + default=42, + min=0, + max=4294967295, + control_after_generate=True, + tooltip="Seed for the generation. Results can still vary between runs " + "with the same seed.", + ), + IO.Autogrow.Input( + "reference_images", + template=IO.Autogrow.TemplateNames( + IO.Image.Input("reference_image"), + names=[f"image_{i}" for i in range(1, 10)], + min=0, + ), + tooltip="Up to 9 images of people, products, or places to use in the " + "video; refer to them as @Image1, @Image2, ...", + ), + IO.Autogrow.Input( + "reference_videos", + template=IO.Autogrow.TemplateNames( + IO.Video.Input("reference_video"), + names=[f"video_{i}" for i in range(1, 4)], + min=0, + ), + tooltip="Up to 3 videos to use as references; refer to them as " + "@Video1, @Video2, ...", + ), + IO.Autogrow.Input( + "reference_audios", + template=IO.Autogrow.TemplateNames( + IO.Audio.Input("reference_audio"), + names=[f"audio_{i}" for i in range(1, 4)], + min=0, + ), + tooltip="Up to 3 audio clips, such as a voice for a speaker; refer to " + "them as @Audio1, @Audio2, ... Requires at least one reference image or " + "video. A voice reference needs a few seconds of clean speech; clips " + "shorter than about 2 seconds are usually ignored.", + ), + ], + ), + ], + ), + ], + outputs=[IO.Video.Output()], + hidden=[ + IO.Hidden.auth_token_comfy_org, + IO.Hidden.api_key_comfy_org, + IO.Hidden.unique_id, + ], + is_api_node=True, + price_badge=IO.PriceBadge( + depends_on=IO.PriceBadgeDepends( + widgets=["model", "model.duration", "model.resolution"], + input_groups=["model.reference_images", "model.reference_videos"], + ), + expr=""" + ( + $dur := $lookup(widgets, "model.duration"); + $hd := $lookup(widgets, "model.resolution") = "768p"; + $imgsRaw := $lookup(inputGroups, "model.reference_images"); + $imgs := $imgsRaw ? $imgsRaw : 0; + $vidsRaw := $lookup(inputGroups, "model.reference_videos"); + $vids := $vidsRaw ? $vidsRaw : 0; + $rate := ($imgs + $vids) > 0 ? ($hd ? 0.0429 : 0.0286) : ($hd ? 0.02145 : 0.0143); + $vids > 0 + ? {"type":"range_usd","min_usd": $rate * $dur, "max_usd": $rate * ($dur + 5 * $vids)} + : {"type":"usd","usd": $rate * $dur} + ) + """, + ), + ) + + @classmethod + async def execute(cls, model: dict) -> IO.NodeOutput: + reference_images = model.get("reference_images", {}) + reference_videos = model.get("reference_videos", {}) + reference_audios = model.get("reference_audios", {}) + if reference_audios and not (reference_images or reference_videos): + raise ValueError("Reference audio requires at least one reference image or video.") + total = len(reference_images) + len(reference_videos) + len(reference_audios) + if total > 12: + raise ValueError(f"At most 12 references can be connected in total; got {total}.") + for key, image in reference_images.items(): + if get_number_of_images(image) != 1: + raise ValueError(f"Reference image input '{key}' must contain exactly one image, not a batch.") + validate_image_aspect_ratio(image, (1, 4), (4, 1), strict=False) + prompt = _rewrite_reference_tags( + model["prompt"], + {"image": len(reference_images), "video": len(reference_videos), "audio": len(reference_audios)}, + ) + validate_string(prompt, strip_whitespace=True, min_length=1, max_length=32000) + payload = { + "model": model["model"], + "mode": "reference_to_video" if reference_images or reference_videos else "text_to_video", + "prompt": prompt, + "duration": model["duration"], + "resolution": model["resolution"], + "seed": model["seed"], + } + if model["aspect_ratio"] != "auto": + payload["aspect_ratio"] = model["aspect_ratio"] + if reference_images: + urls = await upload_images_to_comfyapi( + cls, list(reference_images.values()), max_images=9, mime_type="image/png" + ) + payload["reference_images"] = [{"type": "url", "url": u} for u in urls] + if reference_videos: + payload["reference_videos"] = [ + {"type": "url", "url": await upload_video_to_comfyapi(cls, v)} for v in reference_videos.values() + ] + if reference_audios: + payload["reference_audio"] = [ + { + "type": "url", + "url": await upload_audio_to_comfyapi( + cls, a, container_format="mp3", codec_name="libmp3lame", mime_type="audio/mpeg" + ), + } + for a in reference_audios.values() + ] + video = await _create_and_poll_video(cls, _MODEL_VIDEOS_PATH, payload) + return IO.NodeOutput(await download_url_to_video_output(video["video_url"])) + + +class HeyGenImageToVideoNode(IO.ComfyNode): + + @classmethod + def define_schema(cls) -> IO.Schema: + return IO.Schema( + node_id="HeyGenImageToVideoNode", + display_name="HeyGen Video 1.0 Image to Video", + category="partner/video/HeyGen", + description="Animate an image into a video with synchronized dialogue and sound using " + "HeyGen Video 1.0. The image is used as the first frame.", + inputs=[ + IO.DynamicCombo.Input( + "model", + options=[ + IO.DynamicCombo.Option( + "heygen-video-1", + [ + IO.Image.Input( + "image", + tooltip="First frame of the video. The output keeps the aspect ratio " + "of this image; crop it to change the shape of the video.", + ), + IO.String.Input( + "prompt", + multiline=True, + default="", + tooltip="Description of what happens in the video, including any dialogue.", + ), + IO.Int.Input( + "duration", + default=5, + min=5, + max=15, + step=1, + display_mode=IO.NumberDisplay.slider, + tooltip="Duration of the output video in seconds.", + ), + IO.Combo.Input( + "resolution", + options=["768p", "480p"], + tooltip="Output resolution.", + ), + IO.Int.Input( + "seed", + default=42, + min=0, + max=4294967295, + control_after_generate=True, + tooltip="Seed for the generation. Results can still vary between runs " + "with the same seed.", + ), + ], + ), + ], + ), + ], + outputs=[IO.Video.Output()], + hidden=[ + IO.Hidden.auth_token_comfy_org, + IO.Hidden.api_key_comfy_org, + IO.Hidden.unique_id, + ], + is_api_node=True, + price_badge=IO.PriceBadge( + depends_on=IO.PriceBadgeDepends(widgets=["model", "model.duration", "model.resolution"]), + expr=""" + {"type":"usd","usd": ($lookup(widgets, "model.resolution") = "768p" ? 0.02145 : 0.0143) + * $lookup(widgets, "model.duration")} + """, + ), + ) + + @classmethod + async def execute(cls, model: dict) -> IO.NodeOutput: + validate_string(model["prompt"], strip_whitespace=True, min_length=1, max_length=32000) + if get_number_of_images(model["image"]) != 1: + raise ValueError("The image input must contain exactly one image, not a batch.") + validate_image_aspect_ratio(model["image"], (1, 4), (4, 1), strict=False) + image_url = await upload_image_to_comfyapi(cls, model["image"], mime_type="image/png") + payload = { + "model": model["model"], + "mode": "image_to_video", + "prompt": model["prompt"], + "image": {"type": "url", "url": image_url}, + "duration": model["duration"], + "resolution": model["resolution"], + "seed": model["seed"], + } + video = await _create_and_poll_video(cls, _MODEL_VIDEOS_PATH, payload) + return IO.NodeOutput(await download_url_to_video_output(video["video_url"])) + + class HeyGenExtension(ComfyExtension): @override async def get_node_list(self) -> list[type[IO.ComfyNode]]: @@ -792,6 +1083,8 @@ async def get_node_list(self) -> list[type[IO.ComfyNode]]: HeyGenCreateAvatarNode, HeyGenVideoTranslateNode, HeyGenTextToSpeechNode, + HeyGenReferenceToVideoNode, + HeyGenImageToVideoNode, ] diff --git a/comfy_api_nodes/nodes_ideogram.py b/comfy_api_nodes/nodes_ideogram.py index 37fcd2a0e04..b69bc8966e6 100644 --- a/comfy_api_nodes/nodes_ideogram.py +++ b/comfy_api_nodes/nodes_ideogram.py @@ -1,10 +1,14 @@ +import math +import re from io import BytesIO from typing_extensions import override +from comfy.utils import common_upscale from comfy_api.latest import IO, ComfyExtension from PIL import Image import numpy as np import torch from comfy_api_nodes.apis.ideogram import ( + Ideogram45Request, IdeogramGenerateResponse, IdeogramPImageRequest, IdeogramV3Request, @@ -15,8 +19,10 @@ ApiEndpoint, bytesio_to_image_tensor, download_url_as_bytesio, + download_url_to_image_tensor, resize_mask_to_image, sync_op, + tensor_to_bytesio, validate_string, ) @@ -112,6 +118,56 @@ "1536x640" ] +IDEOGRAM_45_GENERATE_PATH = "/proxy/ideogram/v2/image/generate/ideogram-4-5" +IDEOGRAM_45_PRECISE_EDIT_PATH = "/proxy/ideogram/v2/image/precise-edit/ideogram-4-5" +IDEOGRAM_45_MODELS = ["ideogram-4.5"] +IDEOGRAM_45_MAX_IMAGES = 5 +IDEOGRAM_45_MAX_PIXELS = 4194304 +IDEOGRAM_45_MAX_SIDE = 4608 +IDEOGRAM_45_SIZES = [ + "(2K) 2048x2048 (1:1)", + "(2K) 1440x2880 (1:2)", + "(2K) 2880x1440 (2:1)", + "(2K) 1664x2496 (2:3)", + "(2K) 2496x1664 (3:2)", + "(2K) 1792x2240 (4:5)", + "(2K) 2240x1792 (5:4)", + "(2K) 1440x2560 (9:16)", + "(2K) 2560x1440 (16:9)", + "(2K) 1600x2560 (5:8)", + "(2K) 2560x1600 (8:5)", + "(2K) 1728x2304 (3:4)", + "(2K) 2304x1728 (4:3)", + "(2K) 1296x3168 (9:22)", + "(2K) 3168x1296 (22:9)", + "(2K) 1152x2944 (9:23)", + "(2K) 2944x1152 (23:9)", + "(2K) 1248x3328 (3:8)", + "(2K) 3328x1248 (8:3)", + "(2K) 1280x3072 (5:12)", + "(2K) 3072x1280 (12:5)", + "(2K) 1024x3072 (1:3)", + "(2K) 3072x1024 (3:1)", + "(1K) 1024x1024 (1:1)", + "(1K) 896x1120 (4:5)", + "(1K) 1120x896 (5:4)", + "(1K) 864x1152 (3:4)", + "(1K) 1152x864 (4:3)", + "(1K) 832x1248 (2:3)", + "(1K) 1248x832 (3:2)", + "(1K) 800x1280 (5:8)", + "(1K) 1280x800 (8:5)", + "(1K) 720x1280 (9:16)", + "(1K) 1280x720 (16:9)", + "(1K) 720x1440 (1:2)", + "(1K) 1440x720 (2:1)", +] +IDEOGRAM_45_EDIT_SIZES = [ + s for s in IDEOGRAM_45_SIZES if all(int(v) % 32 == 0 for v in s.split(" ")[1].split("x")) +] +_IMAGE_REF_RE = re.compile(r"@image(?P\d*)(?!\w)", re.IGNORECASE | re.ASCII) + + async def download_and_process_images(image_urls): """Helper function to download and process multiple images from URLs""" @@ -663,6 +719,367 @@ async def execute( ) +def _resolve_image_refs(prompt: str, total_images: int) -> str: + parts = [] + pos = 0 + prev_end = -1 + for match in _IMAGE_REF_RE.finditer(prompt): + start = match.start() + if start > 0 and start != prev_end and (prompt[start - 1].isalnum() or prompt[start - 1] == "_"): + continue + idx = int(match.group("idx") or 1) + if not 1 <= idx <= total_images: + raise ValueError( + f"The prompt references @Image{idx}, but only {total_images} images " + f"are connected (a batched input counts once per image)." + ) + parts.append(prompt[pos:start]) + parts.append(f"image {idx}") + pos = match.end() + prev_end = match.end() + parts.append(prompt[pos:]) + return "".join(parts) + + +def _ideogram_45_images(model: dict) -> list[torch.Tensor]: + images = [image for key in model["images"] for image in model["images"][key]] + if len(images) > IDEOGRAM_45_MAX_IMAGES: + raise ValueError( + f"A maximum of {IDEOGRAM_45_MAX_IMAGES} images is supported; got {len(images)} " + f"(a batched input counts once per image)." + ) + for i, image in enumerate(images, start=1): + height, width = image.shape[0], image.shape[1] + if max(width, height) > 6 * min(width, height): + raise ValueError(f"Image {i} is {width}x{height}; its aspect ratio must be between 1:6 and 6:1.") + return images + + +def _ideogram_45_image_file(image: torch.Tensor) -> BytesIO: + image = image.unsqueeze(0) + height, width = image.shape[1], image.shape[2] + scale = min(1.0, IDEOGRAM_45_MAX_SIDE / max(width, height), math.sqrt(IDEOGRAM_45_MAX_PIXELS / (width * height))) + while True: + new_width, new_height = max(1, round(width * scale)), max(1, round(height * scale)) + if math.ceil(new_width / 32) * math.ceil(new_height / 32) * 1024 <= IDEOGRAM_45_MAX_PIXELS: + break + scale *= 0.995 + if (new_width, new_height) != (width, height): + image = common_upscale(image.movedim(-1, 1), new_width, new_height, "lanczos", "disabled").movedim(1, -1) + return tensor_to_bytesio(image, total_pixels=None, mime_type="image/png") + + +async def _ideogram_45_output(cls: type[IO.ComfyNode], response: IdeogramGenerateResponse) -> torch.Tensor: + data = response.data or [] + urls = [item.url for item in data if item.url] + if not urls: + if any(item.is_image_safe is False for item in data): + raise Exception( + "The result was blocked by Ideogram's content safety filter. " + "Adjust the prompt or images and try again." + ) + raise Exception("No images were generated in the response") + return torch.cat([await download_url_to_image_tensor(url, cls=cls) for url in urls]) + + +def _ideogram_45_quality_input(options: list[str]) -> IO.Combo.Input: + return IO.Combo.Input( + "quality", + options=options, + default="medium", + tooltip="Quality tier. Higher tiers cost more and take longer.", + ) + + +def _ideogram_45_seed_input(tooltip: str) -> IO.Int.Input: + return IO.Int.Input( + "seed", + default=42, + min=0, + max=2147483647, + step=1, + control_after_generate=True, + display_mode=IO.NumberDisplay.number, + tooltip=tooltip, + ) + + +def _ideogram_45_edit_inputs(with_size: bool) -> list: + inputs = [ + IO.Autogrow.Input( + "images", + template=IO.Autogrow.TemplateNames( + IO.Image.Input("image"), + names=[f"image_{i}" for i in range(1, IDEOGRAM_45_MAX_IMAGES + 1)], + min=1, + ), + tooltip="Image 1 is the image to edit; images 2-5 are optional references. " + "Refer to them in the prompt as @Image1, @Image2, ...; a batched input counts once per image.", + ), + IO.String.Input( + "prompt", + multiline=True, + default="", + tooltip="Editing instructions. Supports @Image1-style references to the input images.", + ), + ] + if with_size: + inputs.extend( + [ + IO.Combo.Input( + "size", + options=["auto", "source", *IDEOGRAM_45_EDIT_SIZES, "custom"], + default="auto", + tooltip="Output size. 'auto' picks a ~2K canvas from the images and prompt, 'source' keeps " + "the size of image 1 (images above ~4 MP are scaled down first), and a preset with a different " + "aspect ratio recomposes the scene. Select 'custom' to use the width and height below.", + ), + IO.Int.Input( + "width", + default=2048, + min=256, + max=IDEOGRAM_45_MAX_SIDE, + step=32, + tooltip="Custom output width. Used only when size is set to 'custom'.", + ), + IO.Int.Input( + "height", + default=2048, + min=256, + max=IDEOGRAM_45_MAX_SIDE, + step=32, + tooltip="Custom output height. Used only when size is set to 'custom'.", + ), + ] + ) + inputs.extend( + [ + _ideogram_45_quality_input(["very_low", "low", "medium", "high"]), + _ideogram_45_seed_input("Seed for generation. The same images, prompt, settings and seed give the same result."), + ] + ) + return inputs + + +def _ideogram_45_price_badge() -> IO.PriceBadge: + return IO.PriceBadge( + depends_on=IO.PriceBadgeDepends(widgets=["model", "model.quality"]), + expr=""" + ( + $q := $lookup(widgets, "model.quality"); + {"type": "usd", "usd": $q = "very_low" ? 0.01144 : $q = "low" ? 0.0429 : $q = "high" ? 0.286 : 0.0858} + ) + """, + ) + + +class IdeogramTextToImageApi(IO.ComfyNode): + + @classmethod + def define_schema(cls): + return IO.Schema( + node_id="IdeogramTextToImageApi", + display_name="Ideogram 4.5 Text to Image", + category="partner/image/Ideogram", + description="Generates images from a text prompt using Ideogram 4.5.", + inputs=[ + IO.DynamicCombo.Input( + "model", + options=[ + IO.DynamicCombo.Option( + model_id, + [ + IO.String.Input( + "prompt", + multiline=True, + default="", + tooltip="Text prompt. Also accepts an Ideogram structured JSON caption, " + "for example a previous final_prompt.", + ), + IO.Combo.Input( + "size", + options=["auto", *IDEOGRAM_45_SIZES], + default="auto", + tooltip="Output size. 'auto' lets the model pick a canvas that suits the prompt.", + ), + _ideogram_45_quality_input(["low", "medium", "high"]), + IO.Combo.Input( + "magic_prompt", + options=["auto", "on", "off"], + default="auto", + tooltip="Rewrites the prompt into a detailed structured caption before " + "generating; 'off' keeps your wording as literal as possible. " + "The caption is returned as final_prompt.", + advanced=True, + ), + _ideogram_45_seed_input( + "Seed for generation. Text-to-image is not reproducible from the seed alone " + "because the prompt is rewritten on every run; to reproduce an image, reuse " + "its final_prompt with magic_prompt set to 'off' and the same seed." + ), + ], + ) + for model_id in IDEOGRAM_45_MODELS + ], + tooltip="Model to use.", + ), + ], + outputs=[ + IO.Image.Output(), + IO.String.Output( + "final_prompt", + tooltip="The structured caption the image was generated from. Feed it back with " + "magic_prompt set to 'off' and the same seed to reproduce the image.", + ), + ], + hidden=[ + IO.Hidden.auth_token_comfy_org, + IO.Hidden.api_key_comfy_org, + IO.Hidden.unique_id, + ], + is_api_node=True, + price_badge=_ideogram_45_price_badge(), + ) + + @classmethod + async def execute(cls, model: dict): + validate_string(model["prompt"], strip_whitespace=True, min_length=1, max_length=10000) + response = await sync_op( + cls, + ApiEndpoint(path=IDEOGRAM_45_GENERATE_PATH, method="POST"), + response_model=IdeogramGenerateResponse, + data=Ideogram45Request( + prompt=model["prompt"], + quality=model["quality"], + seed=model["seed"], + size=None if model["size"] == "auto" else model["size"].split(" ")[1], + magic_prompt=model["magic_prompt"], + ), + ) + image = await _ideogram_45_output(cls, response) + return IO.NodeOutput(image, response.data[0].prompt or model["prompt"]) + + +class IdeogramEditApi(IO.ComfyNode): + + @classmethod + def define_schema(cls): + return IO.Schema( + node_id="IdeogramEditApi", + display_name="Ideogram 4.5 Edit", + category="partner/image/Ideogram", + description="Edits or combines up to 5 images guided by a text prompt using Ideogram 4.5. " + "Re-renders the whole image and can change its size or aspect ratio; " + "use Ideogram 4.5 Precise Edit to keep untouched pixels unchanged.", + inputs=[ + IO.DynamicCombo.Input( + "model", + options=[ + IO.DynamicCombo.Option(model_id, _ideogram_45_edit_inputs(with_size=True)) + for model_id in IDEOGRAM_45_MODELS + ], + tooltip="Model to use.", + ), + ], + outputs=[ + IO.Image.Output(), + ], + hidden=[ + IO.Hidden.auth_token_comfy_org, + IO.Hidden.api_key_comfy_org, + IO.Hidden.unique_id, + ], + is_api_node=True, + price_badge=_ideogram_45_price_badge(), + ) + + @classmethod + async def execute(cls, model: dict): + validate_string(model["prompt"], strip_whitespace=True, min_length=1, max_length=10000) + images = _ideogram_45_images(model) + size = model["size"] + if size == "custom": + width, height = model["width"], model["height"] + if width * height > IDEOGRAM_45_MAX_PIXELS: + raise ValueError( + f"Custom size {width}x{height} exceeds the maximum of {IDEOGRAM_45_MAX_PIXELS} pixels (2048x2048)." + ) + if max(width, height) > 6 * min(width, height): + raise ValueError(f"Custom size {width}x{height} exceeds the maximum aspect ratio of 6:1.") + size = f"{width}x{height}" + elif size == "auto": + size = None + elif size != "source": + size = size.split(" ")[1] + prompt = _resolve_image_refs(model["prompt"], len(images)) + response = await sync_op( + cls, + ApiEndpoint(path=IDEOGRAM_45_GENERATE_PATH, method="POST"), + response_model=IdeogramGenerateResponse, + data=Ideogram45Request(prompt=prompt, quality=model["quality"], seed=model["seed"], size=size), + files=[ + ("images", (f"image_{i}.png", _ideogram_45_image_file(image), "image/png")) + for i, image in enumerate(images, start=1) + ], + content_type="multipart/form-data", + ) + return IO.NodeOutput(await _ideogram_45_output(cls, response)) + + +class IdeogramPreciseEditApi(IO.ComfyNode): + + @classmethod + def define_schema(cls): + return IO.Schema( + node_id="IdeogramPreciseEditApi", + display_name="Ideogram 4.5 Precise Edit", + category="partner/image/Ideogram", + description="Edits an image guided by a text prompt using Ideogram 4.5 precise editing: only what the " + "prompt asks for changes, untouched pixels stay identical and the output keeps the size of image 1 " + "(images above ~4 MP are scaled down first). Accepts up to 4 reference images.", + inputs=[ + IO.DynamicCombo.Input( + "model", + options=[ + IO.DynamicCombo.Option(model_id, _ideogram_45_edit_inputs(with_size=False)) + for model_id in IDEOGRAM_45_MODELS + ], + tooltip="Model to use.", + ), + ], + outputs=[ + IO.Image.Output(), + ], + hidden=[ + IO.Hidden.auth_token_comfy_org, + IO.Hidden.api_key_comfy_org, + IO.Hidden.unique_id, + ], + is_api_node=True, + price_badge=_ideogram_45_price_badge(), + ) + + @classmethod + async def execute(cls, model: dict): + validate_string(model["prompt"], strip_whitespace=True, min_length=1, max_length=10000) + images = _ideogram_45_images(model) + prompt = _resolve_image_refs(model["prompt"], len(images)) + files = [("image", ("image_1.png", _ideogram_45_image_file(images[0]), "image/png"))] + files.extend( + ("reference_images", (f"image_{i}.png", _ideogram_45_image_file(image), "image/png")) + for i, image in enumerate(images[1:], start=2) + ) + response = await sync_op( + cls, + ApiEndpoint(path=IDEOGRAM_45_PRECISE_EDIT_PATH, method="POST"), + response_model=IdeogramGenerateResponse, + data=Ideogram45Request(prompt=prompt, quality=model["quality"], seed=model["seed"]), + files=files, + content_type="multipart/form-data", + ) + return IO.NodeOutput(await _ideogram_45_output(cls, response)) + + class IdeogramExtension(ComfyExtension): @override async def get_node_list(self) -> list[type[IO.ComfyNode]]: @@ -670,6 +1087,9 @@ async def get_node_list(self) -> list[type[IO.ComfyNode]]: IdeogramV3, IdeogramV4, IdeogramPImage, + IdeogramTextToImageApi, + IdeogramEditApi, + IdeogramPreciseEditApi, ] diff --git a/comfyui_version.py b/comfyui_version.py index 427334ffc63..ab7337dd71e 100644 --- a/comfyui_version.py +++ b/comfyui_version.py @@ -1,3 +1,3 @@ # This file is automatically generated by the build process when version is # updated in pyproject.toml. -__version__ = "0.38.0" +__version__ = "0.38.1" diff --git a/pyproject.toml b/pyproject.toml index 931b5cd9a5b..bf232f60eb5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "ComfyUI" -version = "0.38.0" +version = "0.38.1" readme = "README.md" license = { file = "LICENSE" } requires-python = ">=3.10" diff --git a/requirements.txt b/requirements.txt index 4dc2c483465..af229a60621 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,5 +1,5 @@ comfyui-frontend-package==1.53.6 -comfyui-workflow-templates==0.11.70 +comfyui-workflow-templates==0.11.74 comfyui-embedded-docs==0.5.12 torch torchsde