Skip to content
Closed
3 changes: 3 additions & 0 deletions comfy/ldm/qwen_image21/model.py
Original file line number Diff line number Diff line change
Expand Up @@ -395,6 +395,9 @@ def block_wrap(args):
hidden_states = block(hidden_states, mod, pe, attn_fn, prefix_len, transformer_options)
for p in patches.get("single_block", []):
hidden_states = p({"img": hidden_states, "x": x, "block_index": i, "transformer_options": transformer_options})["img"]
if cache is not None:
# Release dequantized K/V before leaving the block's allocation scope.
del attn_fn, prefix_k, prefix_v

comfy.model_prefetch.prefetch_queue_pop(prefetch_queue, x.device, None, malloc_scope="block")
comfy.model_prefetch.malloc_graph_end()
Expand Down
2 changes: 1 addition & 1 deletion comfy_api_nodes/apis/anthropic.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,7 +36,7 @@ class AnthropicMessage(BaseModel):


class AnthropicThinkingConfig(BaseModel):
type: Literal["enabled", "disabled", "adaptive"] = Field(...)
type: Literal["enabled", "disabled", "adaptive", "between_tools"] = Field(...)
budget_tokens: int | None = Field(
None, ge=1024,
description="Reasoning budget in tokens. Used when type is 'enabled'. Must be less than max_tokens.",
Expand Down
11 changes: 11 additions & 0 deletions comfy_api_nodes/apis/bfl.py
Original file line number Diff line number Diff line change
Expand Up @@ -184,3 +184,14 @@ class BFLFluxVideoEditRequest(BaseModel):
video: str = Field(..., description="MP4 (URL or base64), at most 15 seconds and 50 MiB.")
prompt: str = Field(..., description="Edit instruction, 1 to 4096 characters once trimmed.")
safety_tolerance: int = Field(4, ge=0, le=4)


class Flux3ImageRequest(BaseModel):
model_config = ConfigDict(extra="forbid")

prompt: str = Field(...)
images: list[str] | None = Field(None, description="1 to 10 reference images (URL or base64).")
aspect_ratio: str = Field("auto")
resolution: str = Field("1k")
grounding: bool = Field(True, description="Web and image search before generating.")
safety_tolerance: int = Field(2, ge=0, le=4)
8 changes: 8 additions & 0 deletions comfy_api_nodes/apis/ideogram.py
Original file line number Diff line number Diff line change
Expand Up @@ -264,3 +264,11 @@ class IdeogramV4Request(BaseModel):
resolution: str | None = Field(None, description="Output resolution in WIDTHxHEIGHT (e.g. '2048x2048').")
rendering_speed: str | None = Field(None, description="Rendering speed: 'TURBO', 'DEFAULT', or 'QUALITY'.")
enable_copyright_detection: bool | None = Field(None, description="Opt into post-generation copyright detection.")


class Ideogram45Request(BaseModel):
prompt: str
quality: str
seed: int = Field(..., ge=0, le=2147483647)
size: str | None = None
magic_prompt: str | None = None
14 changes: 8 additions & 6 deletions comfy_api_nodes/nodes_anthropic.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@
"Opus 4.8": "claude-opus-4-8",
"Fable 5.1": "claude-fable-5-1",
"Fable 5": "claude-fable-5",
"Sonnet 5.5": "claude-sonnet-5-5",
"Sonnet 5": "claude-sonnet-5",
"Opus 4.7": "claude-opus-4-7",
"Opus 4.6": "claude-opus-4-6",
Expand All @@ -44,12 +45,13 @@
_THINKING_UNSUPPORTED = {"Haiku 4.5"}
# Models that use the newer "adaptive" thinking mode (Opus 4.7+ require it; older models keep the explicit budget API).
# Anthropic decides the actual budget when adaptive is used, based on the `output_config.effort` hint.
_ADAPTIVE_THINKING_MODELS = {"Opus 4.8", "Sonnet 5", "Opus 4.7", "Opus 4.6", "Sonnet 4.6"}
_ADAPTIVE_THINKING_MODELS = {"Opus 4.8", "Sonnet 5.5", "Sonnet 5", "Opus 4.7", "Opus 4.6", "Sonnet 4.6"}
_ALWAYS_THINKING_MODELS = {"Opus 5.5", "Opus 5", "Fable 5.1", "Fable 5"}
_XHIGH_EFFORT_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5", "Opus 4.7"}
_XHIGH_EFFORT_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5.5", "Sonnet 5", "Opus 4.7"}
_MAX_EFFORT_MODELS = _XHIGH_EFFORT_MODELS | {"Opus 4.6", "Sonnet 4.6"}
_EXPLICIT_THINKING_OFF_MODELS = {"Sonnet 5"}
_NO_TEMPERATURE_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5"}
_EXPLICIT_THINKING_OFF_MODELS = {"Sonnet 5.5": "between_tools", "Sonnet 5": "disabled"}
_NO_TEMPERATURE_MODELS = {"Opus 5.5", "Opus 5", "Opus 4.8", "Fable 5.1", "Fable 5", "Sonnet 5.5", "Sonnet 5"}
_LOW_MAX_TOKENS_MODELS = {"Opus 5.5", "Sonnet 5.5"}

# Budget mode (Sonnet 4.5): effort -> reasoning budget in tokens. Must be < max_tokens.
# Sized so even the "high" budget fits comfortably under the default max_tokens=32768.
Expand Down Expand Up @@ -77,7 +79,7 @@ def _claude_model_inputs(model_label: str):
IO.Int.Input(
"max_tokens",
default=32768,
min=4096,
min=1024 if model_label in _LOW_MAX_TOKENS_MODELS else 4096,
max=64000,
tooltip="Maximum number of tokens to generate (includes reasoning tokens when enabled).",
advanced=True,
Expand Down Expand Up @@ -297,7 +299,7 @@ async def execute(
budget = min(budget, max(1024, max_tokens - 1024))
thinking_cfg = AnthropicThinkingConfig(type="enabled", budget_tokens=budget)
elif model_label in _EXPLICIT_THINKING_OFF_MODELS:
thinking_cfg = AnthropicThinkingConfig(type="disabled")
thinking_cfg = AnthropicThinkingConfig(type=_EXPLICIT_THINKING_OFF_MODELS[model_label])

image_tensors: list[Input.Image] = [t for t in (images or {}).values() if t is not None]
if sum(get_number_of_images(t) for t in image_tensors) > CLAUDE_MAX_IMAGES:
Expand Down
173 changes: 173 additions & 0 deletions comfy_api_nodes/nodes_bfl.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
import json
import math

import torch
Expand All @@ -18,6 +19,7 @@
BFLFluxVTORequest,
BFLStatus,
Flux2ProGenerateRequest,
Flux3ImageRequest,
Flux3ImageToVideoRequest,
Flux3TextToVideoRequest,
Flux3VideoContinuationRequest,
Expand Down Expand Up @@ -1616,6 +1618,176 @@ async def execute(
return await _bfl_video_execute(cls, _FLUX_VIDEO_EDIT_ENDPOINT, request, poll_via_proxy=True)


_FLUX3_IMAGE_ENDPOINT = ApiEndpoint(path="/proxy/bfl/v1/flux-3-image", method="POST")
_FLUX3_IMAGE_ASPECT_RATIOS = [
"auto", "21:9", "2:1", "16:9", "3:2", "7:5", "4:3", "5:4", "1:1", "4:5", "3:4", "5:7", "2:3", "9:16", "1:2", "9:21"
]
_FLUX3_IMAGE_RESOLUTIONS = {"0.75K": "768sq", "1K": "1k", "1.5K": "1.5k", "2K": "2k", "4K": "4k"}
_FLUX3_IMAGE_MAX_PROMPT_LENGTH = 15000


def _flux3_box_rows(elements: list) -> list[dict]:
rows = []
for index, element in enumerate(elements, start=1):
parts = []
if element.get("type") == "text":
parts.append(f'the text "{element.get("text", "")}"')
if element.get("desc"):
parts.append(element["desc"])
if element.get("color_palette"):
parts.append("colors " + ", ".join(element["color_palette"]))
row = {"id": f"box_{index}", "bbox": element["bbox"]}
if parts:
row["desc"] = ", ".join(parts)
rows.append(row)
return rows
Comment thread
coderabbitai[bot] marked this conversation as resolved.


class Flux3ImageNode(IO.ComfyNode):

@classmethod
def define_schema(cls) -> IO.Schema:
return IO.Schema(
node_id="Flux3ImageNode",
display_name="Flux 3 Image",
category="partner/image/BFL",
description="Generates an image with FLUX 3 from a prompt, or edits and combines up to 10 "
"reference images. Refer to the references in the prompt as image 1, image 2, and so on.",
inputs=[
IO.String.Input(
"prompt",
multiline=True,
default="",
tooltip="What to generate, or the edit to make. The prompt is interpreted and "
"expanded before generation.",
),
IO.Autogrow.Input(
"images",
template=IO.Autogrow.TemplateNames(
IO.Image.Input("image"),
names=[f"image_{i}" for i in range(1, _FLUX3_MAX_IMAGES + 1)],
min=0,
),
tooltip="Optional reference images, up to 10 in total, at least 256x256 pixels each.",
),
IO.Array.Input(
"bounding_boxes",
optional=True,
tooltip="Optional boxes from Create Bounding Boxes that place objects or text in the "
"output. Positions are relative to the canvas, so give it the output's aspect ratio.",
),
IO.Combo.Input(
"aspect_ratio",
options=_FLUX3_IMAGE_ASPECT_RATIOS,
default="auto",
tooltip="'auto' follows the first reference image, or picks a ratio from the prompt.",
),
IO.Combo.Input(
"resolution",
options=list(_FLUX3_IMAGE_RESOLUTIONS),
default="2K",
tooltip="Output size at the chosen aspect ratio: 0.75K is about 0.6 megapixels, "
"1K 1 MP, 1.5K 2.4 MP, 2K 4.2 MP, 4K 16.8 MP.",
),
IO.Boolean.Input(
"grounding",
default=True,
tooltip="Let the model research the prompt with web and image search before generating.",
),
IO.Int.Input(
"safety_tolerance",
default=4,
min=0,
max=4,
advanced=True,
tooltip="Moderation tolerance, 0 is the strictest.",
),
IO.Int.Input(
"seed",
default=42,
min=0,
max=0xFFFFFFFF,
control_after_generate=True,
tooltip="Seed to determine if node should re-run; FLUX 3 picks its own seed, so "
"actual results are nondeterministic regardless of this value.",
),
],
outputs=[IO.Image.Output()],
hidden=[
IO.Hidden.auth_token_comfy_org,
IO.Hidden.api_key_comfy_org,
IO.Hidden.unique_id,
],
is_api_node=True,
price_badge=IO.PriceBadge(
depends_on=IO.PriceBadgeDepends(widgets=["resolution"]),
expr="""
(
$prices := {"0.75k": 0.05863, "1k": 0.06864, "1.5k": 0.1001, "2k": 0.143, "4k": 0.86801};
{"type": "usd", "usd": $lookup($prices, widgets.resolution)}
)
""",
),
)

@classmethod
async def execute(
cls,
prompt: str,
images: IO.Autogrow.Type,
aspect_ratio: str,
resolution: str,
grounding: bool,
safety_tolerance: int,
seed: int,
bounding_boxes: list | None = None,
) -> IO.NodeOutput:
validate_string(prompt, field_name="prompt", min_length=1, max_length=_FLUX3_IMAGE_MAX_PROMPT_LENGTH)
if bounding_boxes:
prompt = f"{prompt} {json.dumps(_flux3_box_rows(bounding_boxes), ensure_ascii=False)}"
if len(prompt) > _FLUX3_IMAGE_MAX_PROMPT_LENGTH:
raise ValueError(
f"The prompt together with the bounding boxes is {len(prompt)} characters long, "
f"the limit is {_FLUX3_IMAGE_MAX_PROMPT_LENGTH}. Shorten the prompt or the box descriptions."
)
reference_images = _flux3_collect_images(images, "reference images")
image_urls = None
if reference_images:
image_urls = await upload_images_to_comfyapi(
cls, reference_images, max_images=_FLUX3_MAX_IMAGES, wait_label="Uploading references"
)
initial_response = await sync_op(
cls,
_FLUX3_IMAGE_ENDPOINT,
response_model=BFLFluxProGenerateResponse,
data=Flux3ImageRequest(
prompt=prompt,
images=image_urls,
aspect_ratio=aspect_ratio,
resolution=_FLUX3_IMAGE_RESOLUTIONS[resolution],
grounding=grounding,
safety_tolerance=safety_tolerance,
),
)
response = await poll_op(
cls,
ApiEndpoint(path=_BFL_POLL_PROXY_PATH, query_params={"polling_url": initial_response.polling_url}),
response_model=BFLFluxStatusResponse,
status_extractor=lambda r: r.status,
progress_extractor=lambda r: r.progress,
completed_statuses=[BFLStatus.ready],
failed_statuses=[
BFLStatus.request_moderated,
BFLStatus.content_moderated,
BFLStatus.error,
BFLStatus.task_not_found,
],
queued_statuses=[BFLStatus.pending],
max_retries_per_poll=3,
)
return IO.NodeOutput(await download_url_to_image_tensor(response.result["sample"]))


class BFLExtension(ComfyExtension):
@override
async def get_node_list(self) -> list[type[IO.ComfyNode]]:
Expand All @@ -1635,6 +1807,7 @@ async def get_node_list(self) -> list[type[IO.ComfyNode]]:
Flux3VideoContinuationNode,
FluxVideoUpscaleNode,
FluxVideoEditNode,
Flux3ImageNode,
]


Expand Down
25 changes: 15 additions & 10 deletions comfy_api_nodes/nodes_grok.py
Original file line number Diff line number Diff line change
Expand Up @@ -624,24 +624,26 @@ def define_schema(cls):
inputs=[
IO.Combo.Input(
"model",
options=["grok-imagine-video", "grok-imagine-video-1.5"],
options=["grok-imagine-video", "grok-imagine-video-1.5", "grok-imagine-video-1.5-lite"],
default="grok-imagine-video-1.5-lite",
tooltip="The model to use for video generation.",
),
IO.String.Input(
"prompt",
multiline=True,
tooltip="Text description of the desired video. "
"Optional for grok-imagine-video-1.5 when an input image is provided.",
"Optional for the grok-imagine-video-1.5 models when an input image is provided.",
),
IO.Combo.Input(
"resolution",
options=["480p", "720p", "1080p"],
tooltip="The resolution of the output video. 1080p is only available for grok-imagine-video-1.5.",
tooltip="The resolution of the output video. 1080p is not available for grok-imagine-video.",
),
IO.Combo.Input(
"aspect_ratio",
options=["auto", "16:9", "4:3", "3:2", "1:1", "2:3", "3:4", "9:16"],
tooltip="The aspect ratio of the output video.",
tooltip="The aspect ratio of the output video. "
"Ignored when an input image is provided; the video follows the image's aspect ratio.",
),
IO.Int.Input(
"duration",
Expand Down Expand Up @@ -682,10 +684,13 @@ def define_schema(cls):
depends_on=IO.PriceBadgeDepends(widgets=["model", "duration", "resolution"], inputs=["image"]),
expr="""
(
$isLite := widgets.model = "grok-imagine-video-1.5-lite";
$is15 := $contains(widgets.model, "1.5");
$rate := $is15
? (widgets.resolution = "1080p" ? 0.25 : (widgets.resolution = "720p" ? 0.14 : 0.08))
: (widgets.resolution = "720p" ? 0.07 : 0.05);
$rate := $isLite
? (widgets.resolution = "1080p" ? 0.14 : (widgets.resolution = "720p" ? 0.03 : 0.02))
: ($is15
? (widgets.resolution = "1080p" ? 0.25 : (widgets.resolution = "720p" ? 0.14 : 0.08))
: (widgets.resolution = "720p" ? 0.07 : 0.05));
$imgCost := $is15 ? 0.01 : 0.002;
$base := $rate * widgets.duration;
$total := inputs.image.connected ? $base + $imgCost : $base;
Expand All @@ -706,14 +711,14 @@ async def execute(
seed: int,
image: Input.Image | None = None,
) -> IO.NodeOutput:
if resolution == "1080p" and model != "grok-imagine-video-1.5":
raise ValueError(f"1080p resolution is only available for grok-imagine-video-1.5, not '{model}'.")
if resolution == "1080p" and model == "grok-imagine-video":
raise ValueError("1080p resolution is not available for grok-imagine-video.")
image_url = None
if image is not None:
if get_number_of_images(image) != 1:
raise ValueError("Only one input image is supported.")
image_url = InputUrlObject(url=f"data:image/png;base64,{tensor_to_base64_string(image)}")
if image is None or model != "grok-imagine-video-1.5":
if image is None or model == "grok-imagine-video":
validate_string(prompt, strip_whitespace=True, min_length=1)
initial_response = await sync_op(
cls,
Expand Down
Loading
Loading