From 69ea58697bb2f05124f5dc7e00ad111f7cfff645 Mon Sep 17 00:00:00 2001 From: comfyanonymous <121283862+comfyanonymous@users.noreply.github.com> Date: Sat, 11 Jul 2026 17:16:40 -0700 Subject: [PATCH 01/13] Try to fix flash attention related issue on AMD. (#14880) --- comfy/ldm/modules/attention.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/comfy/ldm/modules/attention.py b/comfy/ldm/modules/attention.py index 2411aff5c..e6500cff4 100644 --- a/comfy/ldm/modules/attention.py +++ b/comfy/ldm/modules/attention.py @@ -709,7 +709,7 @@ def attention3_sage(q, k, v, heads, mask=None, attn_precision=None, skip_reshape return out try: - @torch.library.custom_op("flash_attention::flash_attn", mutates_args=()) + @torch.library.custom_op("comfy::flash_attn", mutates_args=()) def flash_attn_wrapper(q: torch.Tensor, k: torch.Tensor, v: torch.Tensor, dropout_p: float = 0.0, causal: bool = False, softmax_scale: float = -1.0) -> torch.Tensor: softmax_scale_arg = None if softmax_scale == -1.0 else softmax_scale From 8b099de36acd81acd1afa3b5442951dc847e0a52 Mon Sep 17 00:00:00 2001 From: Gustavo Schneiter Date: Sun, 12 Jul 2026 01:58:25 -0300 Subject: [PATCH 02/13] Fix SaveVideo description: says images, saves video (#14885) --- comfy_extras/nodes_video.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/comfy_extras/nodes_video.py b/comfy_extras/nodes_video.py index d3acc9ad0..3bfd00be4 100644 --- a/comfy_extras/nodes_video.py +++ b/comfy_extras/nodes_video.py @@ -81,7 +81,7 @@ class SaveVideo(io.ComfyNode): display_name="Save Video", category="video", essentials_category="Basics", - description="Saves the input images to your ComfyUI output directory.", + description="Saves the input videos to your ComfyUI output directory.", inputs=[ io.Video.Input("video", tooltip="The video to save."), io.String.Input("filename_prefix", default="video/ComfyUI", tooltip="The prefix for the file to save. This may include formatting information such as %date:yyyy-MM-dd% or %Empty Latent Image.width% to include values from nodes."), From 917faef771a2fd2f14f44af94f17da3d0b2803a3 Mon Sep 17 00:00:00 2001 From: comfyanonymous <121283862+comfyanonymous@users.noreply.github.com> Date: Sun, 12 Jul 2026 09:43:30 -0700 Subject: [PATCH 03/13] Support PID 1.5 models. (#14894) --- comfy/ldm/pixeldit/model.py | 4 ++ comfy/ldm/pixeldit/pid.py | 64 +++++++++++++++---- comfy/model_detection.py | 37 ++++++++++- tests-unit/comfy_test/model_detection_test.py | 52 +++++++++++++++ 4 files changed, 140 insertions(+), 17 deletions(-) diff --git a/comfy/ldm/pixeldit/model.py b/comfy/ldm/pixeldit/model.py index b044b9b29..3b30b9226 100644 --- a/comfy/ldm/pixeldit/model.py +++ b/comfy/ldm/pixeldit/model.py @@ -197,6 +197,9 @@ class PixDiT_T2I(nn.Module): """Hook for subclasses to inject per-block state into the patch stream (e.g. PiD's LQ gate).""" return s + def _pre_pixel_blocks(self, s, **kwargs): + return s + def _forward(self, x, timesteps, context=None, attention_mask=None, transformer_options={}, **kwargs): H_orig, W_orig = x.shape[2], x.shape[3] x = comfy.ldm.common_dit.pad_to_patch_size(x, (self.patch_size, self.patch_size)) @@ -226,6 +229,7 @@ class PixDiT_T2I(nn.Module): s, y_emb = blk(s, y_emb, condition, pos_img, pos_txt, None, transformer_options=transformer_options) s = F.silu(t_emb + s) + s = self._pre_pixel_blocks(s, **kwargs) s_cond = s.view(B * L, self.hidden_size) x_pixels = self.pixel_embedder(x, patch_size=self.patch_size) for blk in self.pixel_blocks: diff --git a/comfy/ldm/pixeldit/pid.py b/comfy/ldm/pixeldit/pid.py index 21b73907a..8590408d9 100644 --- a/comfy/ldm/pixeldit/pid.py +++ b/comfy/ldm/pixeldit/pid.py @@ -13,15 +13,15 @@ from .model import PixDiT_T2I from .modules import precompute_freqs_cis_2d -class SigmaAwareGatePerTokenPerDim(nn.Module): +class SigmaAwareGate(nn.Module): """gate = sigmoid(content_proj(cat[x, lq]) - exp(log_alpha) * sigma); out = x + gate * lq. Trained init gives ~0.88 gate at sigma=0, ~0.05 at sigma=1. """ - def __init__(self, dim: int, dtype=None, device=None, operations=None): + def __init__(self, dim: int, per_token: bool = False, dtype=None, device=None, operations=None): super().__init__() - self.content_proj = operations.Linear(dim * 2, dim, dtype=dtype, device=device) + self.content_proj = operations.Linear(dim * 2, 1 if per_token else dim, dtype=dtype, device=device) self.log_alpha = nn.Parameter(torch.empty((), dtype=dtype, device=device)) def forward(self, x: torch.Tensor, lq: torch.Tensor, sigma: torch.Tensor) -> torch.Tensor: @@ -36,15 +36,15 @@ class SigmaAwareGatePerTokenPerDim(nn.Module): class ResBlock(nn.Module): """Pre-activation ResNet block: GN -> SiLU -> Conv -> GN -> SiLU -> Conv + skip.""" - def __init__(self, channels: int, num_groups: int = 4, dtype=None, device=None, operations=None): + def __init__(self, channels: int, num_groups: int = 4, conv_padding_mode: str = "zeros", dtype=None, device=None, operations=None): super().__init__() self.block = nn.Sequential( operations.GroupNorm(num_groups, channels, dtype=dtype, device=device), nn.SiLU(), - operations.Conv2d(channels, channels, kernel_size=3, padding=1, dtype=dtype, device=device), + operations.Conv2d(channels, channels, kernel_size=3, padding=1, padding_mode=conv_padding_mode, dtype=dtype, device=device), operations.GroupNorm(num_groups, channels, dtype=dtype, device=device), nn.SiLU(), - operations.Conv2d(channels, channels, kernel_size=3, padding=1, dtype=dtype, device=device), + operations.Conv2d(channels, channels, kernel_size=3, padding=1, padding_mode=conv_padding_mode, dtype=dtype, device=device), ) def forward(self, x: torch.Tensor) -> torch.Tensor: @@ -62,9 +62,13 @@ class LQProjection2D(nn.Module): patch_size: int = 16, sr_scale: int = 4, latent_spatial_down_factor: int = 8, + latent_unpatchify_factor: int = 1, num_res_blocks: int = 4, num_outputs: int = 7, interval: int = 2, + conv_padding_mode: str = "zeros", + gate_per_token: bool = False, + pit_output: bool = False, dtype=None, device=None, operations=None, ): super().__init__() @@ -74,34 +78,38 @@ class LQProjection2D(nn.Module): self.patch_size = patch_size self.sr_scale = sr_scale self.latent_spatial_down_factor = latent_spatial_down_factor + self.latent_unpatchify_factor = latent_unpatchify_factor self.num_outputs = num_outputs self.interval = interval - z_to_patch_ratio = (sr_scale * latent_spatial_down_factor) / patch_size + effective_latent_channels = latent_channels // (latent_unpatchify_factor * latent_unpatchify_factor) + effective_spatial_down_factor = latent_spatial_down_factor // latent_unpatchify_factor + z_to_patch_ratio = (sr_scale * effective_spatial_down_factor) / patch_size self.z_to_patch_ratio = z_to_patch_ratio if z_to_patch_ratio >= 1: self.latent_fold_factor = 0 - latent_proj_in_ch = latent_channels + latent_proj_in_ch = effective_latent_channels else: fold_factor = int(1 / z_to_patch_ratio) assert fold_factor * z_to_patch_ratio == 1.0 self.latent_fold_factor = fold_factor - latent_proj_in_ch = latent_channels * fold_factor * fold_factor + latent_proj_in_ch = effective_latent_channels * fold_factor * fold_factor layers = [ - operations.Conv2d(latent_proj_in_ch, hidden_dim, kernel_size=3, padding=1, dtype=dtype, device=device), + operations.Conv2d(latent_proj_in_ch, hidden_dim, kernel_size=3, padding=1, padding_mode=conv_padding_mode, dtype=dtype, device=device), nn.SiLU(), - operations.Conv2d(hidden_dim, hidden_dim, kernel_size=3, padding=1, dtype=dtype, device=device), + operations.Conv2d(hidden_dim, hidden_dim, kernel_size=3, padding=1, padding_mode=conv_padding_mode, dtype=dtype, device=device), ] for _ in range(num_res_blocks): - layers.append(ResBlock(hidden_dim, dtype=dtype, device=device, operations=operations)) + layers.append(ResBlock(hidden_dim, conv_padding_mode=conv_padding_mode, dtype=dtype, device=device, operations=operations)) self.latent_proj = nn.Sequential(*layers) self.output_heads = nn.ModuleList( [operations.Linear(hidden_dim, out_dim, dtype=dtype, device=device) for _ in range(num_outputs)] ) + self.pit_head = operations.Linear(hidden_dim, out_dim, dtype=dtype, device=device) if pit_output else None self.gate_modules = nn.ModuleList( - [SigmaAwareGatePerTokenPerDim(out_dim, dtype=dtype, device=device, operations=operations) + [SigmaAwareGate(out_dim, per_token=gate_per_token, dtype=dtype, device=device, operations=operations) for _ in range(num_outputs)] ) @@ -115,6 +123,11 @@ class LQProjection2D(nn.Module): return self.gate_modules[out_idx](x, lq_feature, sigma) def _align_latent_to_patch_grid(self, lq_latent: torch.Tensor, pH: int, pW: int) -> torch.Tensor: + f = self.latent_unpatchify_factor + if f > 1: + B, C, H, W = lq_latent.shape + lq_latent = lq_latent.reshape(B, C // (f * f), f, f, H, W) + lq_latent = lq_latent.permute(0, 1, 4, 2, 5, 3).reshape(B, C // (f * f), H * f, W * f) B, z_dim = lq_latent.shape[:2] if self.z_to_patch_ratio >= 1: if lq_latent.shape[2] != pH or lq_latent.shape[3] != pW: @@ -134,7 +147,10 @@ class LQProjection2D(nn.Module): feat = self._align_latent_to_patch_grid(lq_latent, target_pH, target_pW) B, C, H, W = feat.shape tokens = feat.permute(0, 2, 3, 1).contiguous().view(B, H * W, C) - return [head(tokens) for head in self.output_heads] + outputs = [head(tokens) for head in self.output_heads] + if self.pit_head is not None: + outputs.append(self.pit_head(tokens)) + return outputs class PidNet(PixDiT_T2I): @@ -148,6 +164,10 @@ class PidNet(PixDiT_T2I): lq_interval: int = 2, sr_scale: int = 4, latent_spatial_down_factor: int = 8, + lq_latent_unpatchify_factor: int = 1, + lq_conv_padding_mode: str = "zeros", + lq_gate_per_token: bool = False, + pit_lq_inject: bool = False, rope_ref_h: int = 1024, # NTK ref resolution in PIXEL units: 1024px / patch=16 -> grid_ref=64. rope_ref_w: int = 1024, image_model=None, @@ -165,6 +185,8 @@ class PidNet(PixDiT_T2I): for blk in self.pixel_blocks: blk._rope_fn = _pit_rope_fn + self.pit_lq_inject = pit_lq_inject + num_lq_outputs = (self.patch_depth + lq_interval - 1) // lq_interval self.lq_proj = LQProjection2D( latent_channels=lq_latent_channels, @@ -173,13 +195,20 @@ class PidNet(PixDiT_T2I): patch_size=self.patch_size, sr_scale=sr_scale, latent_spatial_down_factor=latent_spatial_down_factor, + latent_unpatchify_factor=lq_latent_unpatchify_factor, num_res_blocks=lq_num_res_blocks, num_outputs=num_lq_outputs, interval=lq_interval, + conv_padding_mode=lq_conv_padding_mode, + gate_per_token=lq_gate_per_token, + pit_output=pit_lq_inject, dtype=dtype, device=device, operations=operations, ) + self.pit_lq_gate = SigmaAwareGate( + self.hidden_size, per_token=lq_gate_per_token, dtype=dtype, device=device, operations=operations + ) if pit_lq_inject else None def _fetch_patch_pos(self, height, width, device, dtype, **rope_opts): return precompute_freqs_cis_2d( @@ -197,6 +226,11 @@ class PidNet(PixDiT_T2I): return s return self.lq_proj.gate(s, pid_lq_features[out_idx], pid_degrade_sigma, out_idx) + def _pre_pixel_blocks(self, s, pid_pit_lq_feature=None, pid_degrade_sigma=None, **kwargs): + if pid_pit_lq_feature is None: + return s + return self.pit_lq_gate(s, pid_pit_lq_feature, pid_degrade_sigma) + def _forward(self, x, timesteps, context=None, attention_mask=None, transformer_options={}, lq_latent=None, degrade_sigma=None, **kwargs): if lq_latent is None: raise ValueError("PidNet requires lq_latent — attach via PiDConditioning") @@ -216,12 +250,14 @@ class PidNet(PixDiT_T2I): degrade_sigma = degrade_sigma.expand(B).contiguous() lq_features = self.lq_proj(lq_latent=lq_latent.to(x), target_pH=Hs, target_pW=Ws) + pit_lq_feature = lq_features.pop() if self.pit_lq_inject else None return super()._forward( x, timesteps, context=context, attention_mask=attention_mask, transformer_options=transformer_options, pid_lq_features=lq_features, + pid_pit_lq_feature=pit_lq_feature, pid_degrade_sigma=degrade_sigma, **kwargs, ) diff --git a/comfy/model_detection.py b/comfy/model_detection.py index 174bc77cc..70c8625e3 100644 --- a/comfy/model_detection.py +++ b/comfy/model_detection.py @@ -470,15 +470,46 @@ def detect_unet_config(state_dict, key_prefix, metadata=None): # PiD (Pixel Diffusion Decoder). Must check BEFORE plain PixelDiT_T2I. _lq_w_key = '{}lq_proj.latent_proj.0.weight'.format(key_prefix) if _lq_w_key in state_dict_keys: - in_ch = int(state_dict[_lq_w_key].shape[1]) + latent_proj_in_channels = int(state_dict[_lq_w_key].shape[1]) + hidden_dim = int(state_dict[_lq_w_key].shape[0]) _gate_prefix = '{}lq_proj.gate_modules.'.format(key_prefix) num_gates = len({k[len(_gate_prefix):].split('.')[0] for k in state_dict_keys if k.startswith(_gate_prefix)}) + pid_v1_5 = '{}lq_proj.pit_head.weight'.format(key_prefix) in state_dict_keys dit_config = {"image_model": "pid", - "lq_latent_channels": in_ch, - "latent_spatial_down_factor": 16 if in_ch >= 64 else 8} + "lq_hidden_dim": hidden_dim} if num_gates > 0: dit_config["lq_interval"] = (14 + num_gates - 1) // num_gates + if pid_v1_5: + pid_v1_5_variants = { + 16: { # Flux and QwenImage + "lq_latent_channels": 16, + "latent_spatial_down_factor": 8, + "lq_latent_unpatchify_factor": 1, + }, + 32: { # Flux2 after 2x latent unpatchify + "lq_latent_channels": 128, + "latent_spatial_down_factor": 16, + "lq_latent_unpatchify_factor": 2, + }, + } + variant = pid_v1_5_variants.get(latent_proj_in_channels) + if variant is None: + raise ValueError(f"Unsupported PiD v1.5 latent projection with {latent_proj_in_channels} input channels") + gate_weight = state_dict['{}lq_proj.gate_modules.0.content_proj.weight'.format(key_prefix)] + dit_config.update(variant) + dit_config.update({ + "lq_conv_padding_mode": "replicate", + "lq_gate_per_token": gate_weight.shape[0] == 1, + "pit_lq_inject": True, + "rope_ref_h": 2048, + "rope_ref_w": 2048, + }) + else: + dit_config.update({ + "lq_latent_channels": latent_proj_in_channels, + "latent_spatial_down_factor": 16 if latent_proj_in_channels >= 64 else 8, + }) return dit_config if '{}core.pixel_embedder.proj.weight'.format(key_prefix) in state_dict_keys: # PixelDiT T2I diff --git a/tests-unit/comfy_test/model_detection_test.py b/tests-unit/comfy_test/model_detection_test.py index 6e7d71f79..7c5b271c5 100644 --- a/tests-unit/comfy_test/model_detection_test.py +++ b/tests-unit/comfy_test/model_detection_test.py @@ -97,6 +97,21 @@ def _make_seedvr2_3b_shared_mm_sd(): } +def _make_pid_v1_5_sd(latent_proj_channels=16): + sd = { + "pixel_embedder.proj.weight": torch.empty(16, 3, device="meta"), + "lq_proj.latent_proj.0.weight": torch.empty(1024, latent_proj_channels, 3, 3, device="meta"), + "lq_proj.pit_head.weight": torch.empty(1536, 1024, device="meta"), + "lq_proj.gate_modules.0.content_proj.weight": torch.empty(1, 3072, device="meta"), + "pixel_blocks.0.attn.q_norm.weight": torch.empty(72, device="meta"), + "pixel_blocks.0.adaLN_modulation.0.weight": torch.empty(24576, 1536, device="meta"), + "pixel_blocks.0.adaLN_modulation.0.bias": torch.empty(24576, device="meta"), + } + for i in range(7): + sd[f"lq_proj.gate_modules.{i}.log_alpha"] = torch.empty((), device="meta") + return sd + + def _add_model_diffusion_prefix(sd): return {f"model.diffusion_model.{k}": v for k, v in sd.items()} @@ -206,6 +221,43 @@ class TestModelDetection: assert type(model_config_from_unet(sd, "model.diffusion_model.")).__name__ == "SeedVR2" + def test_pid_v1_5_detection(self): + sd = _make_pid_v1_5_sd() + unet_config = detect_unet_config(sd, "") + + assert unet_config == { + "image_model": "pid", + "lq_latent_channels": 16, + "lq_hidden_dim": 1024, + "latent_spatial_down_factor": 8, + "lq_interval": 2, + "lq_latent_unpatchify_factor": 1, + "lq_conv_padding_mode": "replicate", + "lq_gate_per_token": True, + "pit_lq_inject": True, + "rope_ref_h": 2048, + "rope_ref_w": 2048, + } + assert type(model_config_from_unet_config(unet_config, sd)).__name__ == "PiD" + + def test_pid_v1_5_flux2_detection(self): + unet_config = detect_unet_config(_make_pid_v1_5_sd(latent_proj_channels=32), "") + + assert unet_config["lq_latent_channels"] == 128 + assert unet_config["latent_spatial_down_factor"] == 16 + assert unet_config["lq_latent_unpatchify_factor"] == 2 + + def test_pid_v1_5_pixel_adaln_conversion(self): + sd = _make_pid_v1_5_sd() + model_config = model_config_from_unet_config(detect_unet_config(sd, ""), sd) + processed = model_config.process_unet_state_dict(sd) + + assert processed["pixel_blocks.0.attn.q_norm.weight"].shape == (72,) + assert processed["pixel_blocks.0.adaLN_modulation_msa.weight"].shape == (12288, 1536) + assert processed["pixel_blocks.0.adaLN_modulation_mlp.weight"].shape == (12288, 1536) + assert processed["pixel_blocks.0.adaLN_modulation_msa.bias"].shape == (12288,) + assert processed["pixel_blocks.0.adaLN_modulation_mlp.bias"].shape == (12288,) + def test_unet_config_and_required_keys_combination_is_unique(self): """Each model in the registry must have a unique combination of ``unet_config`` and ``required_keys``. If two models share the same From b58f829b570d52fbbd41ebb00c6fe02b8755ec75 Mon Sep 17 00:00:00 2001 From: Alexander Piskun <13381981+bigcat88@users.noreply.github.com> Date: Mon, 13 Jul 2026 10:38:35 +0300 Subject: [PATCH 04/13] [Partner Nodes] feat(client): send ComfyUI Core version in request headers (#14910) Signed-off-by: bigcat88 --- comfy_api_nodes/util/_helpers.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/comfy_api_nodes/util/_helpers.py b/comfy_api_nodes/util/_helpers.py index 7eb1ec664..acab10d95 100644 --- a/comfy_api_nodes/util/_helpers.py +++ b/comfy_api_nodes/util/_helpers.py @@ -15,6 +15,7 @@ from comfy.comfy_api_env import normalize_comfy_api_base from comfy.deploy_environment import get_deploy_environment from comfy.model_management import processing_interrupted from comfy_api.latest import IO +from comfyui_version import __version__ as comfyui_version from .common_exceptions import ProcessingInterrupted @@ -60,6 +61,7 @@ def get_comfy_api_headers(node_cls: type[IO.ComfyNode]) -> dict[str, str]: **get_auth_header(node_cls), "Comfy-Env": get_deploy_environment(), "Comfy-Usage-Source": get_usage_source(node_cls), + "Comfy-Core-Version": comfyui_version, } From 8deaa4d911497f93bbd434a3821efab396f6981f Mon Sep 17 00:00:00 2001 From: Alexander Piskun <13381981+bigcat88@users.noreply.github.com> Date: Mon, 13 Jul 2026 10:53:37 +0300 Subject: [PATCH 05/13] fix(image): correct HLG inverse-OETF clamp in hlg_to_linear (#14762) Signed-off-by: bigcat88 Co-authored-by: Alexis Rolland --- comfy_extras/nodes_images.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/comfy_extras/nodes_images.py b/comfy_extras/nodes_images.py index fe1937ba5..4d7b37200 100644 --- a/comfy_extras/nodes_images.py +++ b/comfy_extras/nodes_images.py @@ -891,10 +891,11 @@ def hlg_to_linear(t: torch.Tensor) -> torch.Tensor: return torch.cat([hlg_to_linear(rgb), alpha], dim=-1) # Piecewise: sqrt branch below 0.5, log branch above. - # Clamp inside the log branch so negative / out-of-range values don't blow up; + # Clamp the log branch at the 0.5 branch point (not above it) so the + # unselected lane stays finite in exp() without altering selected values; # values above 1.0 are allowed and extrapolate naturally. low = (t ** 2) / 3.0 - high = (torch.exp((t.clamp(min=_HLG_C) - _HLG_C) / _HLG_A) + _HLG_B) / 12.0 + high = (torch.exp((t.clamp(min=0.5) - _HLG_C) / _HLG_A) + _HLG_B) / 12.0 return torch.where(t <= 0.5, low, high) From ec0e8b3447d5aa5a91a5a846b7fd94c88318fef7 Mon Sep 17 00:00:00 2001 From: Alexander Piskun <13381981+bigcat88@users.noreply.github.com> Date: Mon, 13 Jul 2026 11:38:50 +0300 Subject: [PATCH 06/13] fix(image): support single-channel images in Save Image (Advanced) (#14761) Signed-off-by: bigcat88 Co-authored-by: Alexis Rolland --- comfy_extras/nodes_images.py | 32 +++++++++++++++++++++----------- 1 file changed, 21 insertions(+), 11 deletions(-) diff --git a/comfy_extras/nodes_images.py b/comfy_extras/nodes_images.py index 4d7b37200..7011d9c13 100644 --- a/comfy_extras/nodes_images.py +++ b/comfy_extras/nodes_images.py @@ -844,15 +844,18 @@ class ImageMergeTileList(IO.ComfyNode): # Format specifications # --------------------------------------------------------------------------- -# Maps (file_format, bit_depth, has_alpha) -> (numpy dtype scale, av pixel format, -# stream pix_fmt). Keeps the encode path declarative instead of branchy. +# Maps (file_format, bit_depth, num_channels) -> (quantization scale, numpy dtype, +# av frame pix_fmt, stream pix_fmt). Keeps the encode path declarative instead of branchy. _FORMAT_SPECS = { - ("png", "8-bit", False): {"scale": 255.0, "dtype": np.uint8, "frame_fmt": "rgb24", "stream_fmt": "rgb24"}, - ("png", "8-bit", True): {"scale": 255.0, "dtype": np.uint8, "frame_fmt": "rgba", "stream_fmt": "rgba"}, - ("png", "16-bit", False): {"scale": 65535.0, "dtype": np.uint16, "frame_fmt": "rgb48le", "stream_fmt": "rgb48be"}, - ("png", "16-bit", True): {"scale": 65535.0, "dtype": np.uint16, "frame_fmt": "rgba64le", "stream_fmt": "rgba64be"}, - ("exr", "32-bit float", False): {"scale": 1.0, "dtype": np.float32, "frame_fmt": "gbrpf32le", "stream_fmt": "gbrpf32le"}, - ("exr", "32-bit float", True): {"scale": 1.0, "dtype": np.float32, "frame_fmt": "gbrapf32le", "stream_fmt": "gbrapf32le"}, + ("png", "8-bit", 1): {"scale": 255.0, "dtype": np.uint8, "frame_fmt": "gray", "stream_fmt": "gray"}, + ("png", "8-bit", 3): {"scale": 255.0, "dtype": np.uint8, "frame_fmt": "rgb24", "stream_fmt": "rgb24"}, + ("png", "8-bit", 4): {"scale": 255.0, "dtype": np.uint8, "frame_fmt": "rgba", "stream_fmt": "rgba"}, + ("png", "16-bit", 1): {"scale": 65535.0, "dtype": np.uint16, "frame_fmt": "gray16le", "stream_fmt": "gray16be"}, + ("png", "16-bit", 3): {"scale": 65535.0, "dtype": np.uint16, "frame_fmt": "rgb48le", "stream_fmt": "rgb48be"}, + ("png", "16-bit", 4): {"scale": 65535.0, "dtype": np.uint16, "frame_fmt": "rgba64le", "stream_fmt": "rgba64be"}, + ("exr", "32-bit float", 1): {"scale": 1.0, "dtype": np.float32, "frame_fmt": "grayf32le", "stream_fmt": "grayf32le"}, + ("exr", "32-bit float", 3): {"scale": 1.0, "dtype": np.float32, "frame_fmt": "gbrpf32le", "stream_fmt": "gbrpf32le"}, + ("exr", "32-bit float", 4): {"scale": 1.0, "dtype": np.float32, "frame_fmt": "gbrapf32le", "stream_fmt": "gbrapf32le"}, } @@ -1088,7 +1091,8 @@ def _encode_image( bit_depth: str, colorspace: str, ) -> bytes: - """Encode a single HxWxC tensor to PNG or EXR bytes in memory. + """Encode a single HxWxC (or channel-less HxW grayscale) tensor to PNG or + EXR bytes in memory. Grayscale is written as single-channel PNG / Y-only EXR. For EXR the input is interpreted according to `colorspace` and converted to scene-linear (EXR's convention) before writing: @@ -1102,10 +1106,16 @@ def _encode_image( For PNG, colorspace selection does not modify pixels — PNG is delivered sRGB-encoded and there is no PNG path for wide-gamut HDR in this node. """ + if img_tensor.ndim == 2: + img_tensor = img_tensor.unsqueeze(-1) # Some nodes emit grayscale as (H, W) with no channel dim, mask-style. height, width, num_channels = img_tensor.shape - has_alpha = num_channels == 4 - spec = _FORMAT_SPECS[(file_format, bit_depth, has_alpha)] + spec = _FORMAT_SPECS.get((file_format, bit_depth, num_channels)) + if spec is None: + raise ValueError( + f"No {file_format}/{bit_depth} encoder for {num_channels}-channel images: " + "supported channel counts are 1 (grayscale), 3 (RGB) and 4 (RGBA)." + ) if spec["dtype"] == np.float32: # EXR path: preserve full range, no clamp. From 5697b970173bc0c16a05c30d509d0911f2b84822 Mon Sep 17 00:00:00 2001 From: Alexander Piskun <13381981+bigcat88@users.noreply.github.com> Date: Mon, 13 Jul 2026 14:18:09 +0300 Subject: [PATCH 07/13] [Partner Nodes] chore(Google): reroute Gemini Image preview models to release versions (#14917) Signed-off-by: bigcat88 --- comfy_api_nodes/nodes_gemini.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/comfy_api_nodes/nodes_gemini.py b/comfy_api_nodes/nodes_gemini.py index aa992802d..a8eb0a797 100644 --- a/comfy_api_nodes/nodes_gemini.py +++ b/comfy_api_nodes/nodes_gemini.py @@ -1133,7 +1133,9 @@ class GeminiImage2(IO.ComfyNode): ) -> IO.NodeOutput: validate_string(prompt, strip_whitespace=True, min_length=1) if model == "Nano Banana 2 (Gemini 3.1 Flash Image)": - model = "gemini-3.1-flash-image-preview" + model = "gemini-3.1-flash-image" + elif model == "gemini-3-pro-image-preview": + model = "gemini-3-pro-image" parts: list[GeminiPart] = [GeminiPart(text=prompt)] if images is not None: @@ -1507,7 +1509,7 @@ class GeminiNanoBanana2V2(IO.ComfyNode): validate_string(prompt, strip_whitespace=True, min_length=1) model_choice = model["model"] if model_choice == "Nano Banana 2 (Gemini 3.1 Flash Image)": - model_id = "gemini-3.1-flash-image-preview" + model_id = "gemini-3.1-flash-image" elif model_choice == "Nano Banana 2 Lite": model_id = "gemini-3.1-flash-lite-image" else: From 5bb831a3f565dbcd517fc157284ce69af528ec2e Mon Sep 17 00:00:00 2001 From: "Daxiong (Lin)" Date: Mon, 13 Jul 2026 22:13:23 +0800 Subject: [PATCH 08/13] chore: update embedded docs to v0.5.8 (#14920) --- requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements.txt b/requirements.txt index 790ef4940..b27de8987 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,6 +1,6 @@ comfyui-frontend-package==1.45.20 comfyui-workflow-templates==0.11.6 -comfyui-embedded-docs==0.5.7 +comfyui-embedded-docs==0.5.8 torch torchsde torchvision From 5658a68a875e4c1210d72c71d61eca95740adaf0 Mon Sep 17 00:00:00 2001 From: Comfy Org PR Bot Date: Tue, 14 Jul 2026 02:20:58 +0900 Subject: [PATCH 09/13] chore(openapi): sync shared API contract from cloud@bcb8f5f (#14815) Co-authored-by: mattmillerai <7741082+mattmillerai@users.noreply.github.com> Co-authored-by: Alexis Rolland Co-authored-by: Matt Miller --- openapi.yaml | 45 ++++++++++++++++++++++++--------------------- 1 file changed, 24 insertions(+), 21 deletions(-) diff --git a/openapi.yaml b/openapi.yaml index c09b1eeac..e00643bad 100644 --- a/openapi.yaml +++ b/openapi.yaml @@ -7,18 +7,18 @@ components: description: Timestamp when the asset was created format: date-time type: string + display_name: + description: Display name of the asset. Mirrors name for backwards compatibility. + nullable: true + type: string + file_path: + description: Relative path in global-namespace-root form (e.g. "models/checkpoints/flux.safetensors") + nullable: true + type: string hash: description: Blake3 hash of the asset content. pattern: ^blake3:[a-f0-9]{64}$ type: string - loader_path: - description: The value a loader consumes to load this asset. Null when no loader can resolve the file. - nullable: true - type: string - display_name: - description: Human-facing label for the asset. Not unique. - nullable: true - type: string id: description: Unique identifier for the asset format: uuid @@ -144,6 +144,14 @@ components: AssetUpdated: description: Response returned when an existing asset is successfully updated. properties: + display_name: + description: Display name of the asset. Mirrors name for backwards compatibility. + nullable: true + type: string + file_path: + description: Relative path in global-namespace-root form (e.g. "models/checkpoints/flux.safetensors") + nullable: true + type: string hash: description: Blake3 hash of the asset content. pattern: ^blake3:[a-f0-9]{64}$ @@ -775,14 +783,6 @@ components: ModelFolder: description: Represents a folder containing models properties: - extensions: - description: The folder's registered file-extension allowlist. An empty array means the folder accepts any extension (match-all). - example: - - .ckpt - - .safetensors - items: - type: string - type: array folders: description: List of paths where models of this type are stored example: @@ -1644,7 +1644,7 @@ paths: format: uuid type: string tags: - description: JSON-encoded array of tag strings. For new byte uploads, include exactly one destination role (`input`, `output`, or `models`); `models` uploads also require exactly one `model_type:` tag. Extra tags are stored as labels and do not create path components. + description: JSON-encoded array of freeform tag strings, e.g. '["models","checkpoint"]'. Common types include "models", "input", "output", and "temp", but any tag can be used in any order. type: string user_metadata: description: Custom JSON metadata as a string @@ -1829,7 +1829,7 @@ paths: content: application/json: schema: - $ref: '#/components/schemas/Asset' + $ref: '#/components/schemas/AssetUpdated' description: Asset updated successfully "400": content: @@ -2470,9 +2470,6 @@ paths: supports_preview_metadata: description: Whether the server supports preview metadata type: boolean - supports_model_type_tags: - description: Whether the server supports namespaced model type asset tags - type: boolean type: object description: Success headers: @@ -3300,6 +3297,12 @@ paths: schema: $ref: '#/components/schemas/ErrorResponse' description: Invalid request parameters + "401": + content: + application/json: + schema: + $ref: '#/components/schemas/ErrorResponse' + description: Unauthorized - Authentication required "500": content: application/json: From da2608926eaf68fd532bba4e1ace3402c5d21399 Mon Sep 17 00:00:00 2001 From: "Daxiong (Lin)" Date: Tue, 14 Jul 2026 03:15:34 +0800 Subject: [PATCH 10/13] Update workflow templates to v0.11.9 (#14924) --- requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements.txt b/requirements.txt index b27de8987..e7e7ba747 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,5 +1,5 @@ comfyui-frontend-package==1.45.20 -comfyui-workflow-templates==0.11.6 +comfyui-workflow-templates==0.11.9 comfyui-embedded-docs==0.5.8 torch torchsde From c35a622acdabf99f60bba737f07977fef1ef97f4 Mon Sep 17 00:00:00 2001 From: comfyanonymous <121283862+comfyanonymous@users.noreply.github.com> Date: Mon, 13 Jul 2026 12:52:28 -0700 Subject: [PATCH 11/13] Fix hidream o1 regression. (#14923) --- comfy/ldm/hidream_o1/attention.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/comfy/ldm/hidream_o1/attention.py b/comfy/ldm/hidream_o1/attention.py index 1b68f1771..afb2be9b8 100644 --- a/comfy/ldm/hidream_o1/attention.py +++ b/comfy/ldm/hidream_o1/attention.py @@ -15,24 +15,24 @@ def make_two_pass_attention(ar_len: int, transformer_options=None): The AR pass goes through SDPA directand bypasses wrappers, it is only ~1% of T at typical edit sizes. """ - def two_pass_attention(q, k, v, heads, **kwargs): + def two_pass_attention(q, k, v, heads, enable_gqa=False, **kwargs): B, H, T, D = q.shape if T < k.shape[2]: # KV-cache hot path: Q is shorter than K/V (cached AR prefix is in K/V only), all fresh Q positions are in the gen region, single full-attention call - out = optimized_attention(q, k, v, heads, mask=None, skip_reshape=True, skip_output_reshape=True, transformer_options=transformer_options) + out = optimized_attention(q, k, v, heads, mask=None, skip_reshape=True, skip_output_reshape=True, transformer_options=transformer_options, enable_gqa=enable_gqa) elif ar_len >= T: - out = comfy.ops.scaled_dot_product_attention(q, k, v, attn_mask=None, dropout_p=0.0, is_causal=True) + out = comfy.ops.scaled_dot_product_attention(q, k, v, attn_mask=None, dropout_p=0.0, is_causal=True, enable_gqa=enable_gqa) elif ar_len <= 0: - out = optimized_attention(q, k, v, heads, mask=None, skip_reshape=True, skip_output_reshape=True, transformer_options=transformer_options) + out = optimized_attention(q, k, v, heads, mask=None, skip_reshape=True, skip_output_reshape=True, transformer_options=transformer_options, enable_gqa=enable_gqa) else: out_ar = comfy.ops.scaled_dot_product_attention( q[:, :, :ar_len], k[:, :, :ar_len], v[:, :, :ar_len], - attn_mask=None, dropout_p=0.0, is_causal=True, + attn_mask=None, dropout_p=0.0, is_causal=True, enable_gqa=enable_gqa, ) out_gen = optimized_attention( q[:, :, ar_len:], k, v, heads, mask=None, skip_reshape=True, skip_output_reshape=True, - transformer_options=transformer_options, + transformer_options=transformer_options, enable_gqa=enable_gqa, ) out = torch.cat([out_ar, out_gen], dim=2) From 80acfcf0fe6ea006f0d6176be8301ade1c0c4836 Mon Sep 17 00:00:00 2001 From: comfyanonymous <121283862+comfyanonymous@users.noreply.github.com> Date: Mon, 13 Jul 2026 13:03:29 -0700 Subject: [PATCH 12/13] More optimized int8 and int4 on turing. (#14927) --- requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements.txt b/requirements.txt index e7e7ba747..e1458ca34 100644 --- a/requirements.txt +++ b/requirements.txt @@ -22,7 +22,7 @@ alembic SQLAlchemy>=2.0.0 filelock av>=16.0.0 -comfy-kitchen==0.2.18 +comfy-kitchen==0.2.19 comfy-aimdo==0.4.10 requests simpleeval>=1.0.0 From 0aecac867d7840b56ad790aa76c5e76e33c74c3d Mon Sep 17 00:00:00 2001 From: Terry Jia Date: Mon, 13 Jul 2026 22:07:23 -0400 Subject: [PATCH 13/13] Fix 3D advanced nodes crashing in API mode (no UI) (#14930) --- comfy_extras/nodes_load_3d.py | 12 ++++++++---- comfy_extras/nodes_save_3d.py | 3 ++- 2 files changed, 10 insertions(+), 5 deletions(-) diff --git a/comfy_extras/nodes_load_3d.py b/comfy_extras/nodes_load_3d.py index a9df557c2..106b01f9d 100644 --- a/comfy_extras/nodes_load_3d.py +++ b/comfy_extras/nodes_load_3d.py @@ -174,8 +174,9 @@ class Preview3DAdvanced(IO.ComfyNode): filename = f"preview3d_advanced_{uuid.uuid4().hex}.{model_3d.format}" model_3d.save_to(os.path.join(folder_paths.get_temp_directory(), filename)) + viewport_state = viewport_state if isinstance(viewport_state, dict) else {} camera_info_input = kwargs.get("camera_info", None) - camera_info = camera_info_input if camera_info_input is not None else viewport_state['camera_info'] + camera_info = camera_info_input if camera_info_input is not None else viewport_state.get('camera_info') model_3d_info_input = kwargs.get("model_3d_info", None) model_3d_info = model_3d_info_input if model_3d_info_input is not None else viewport_state.get('model_3d_info', []) return IO.NodeOutput( @@ -243,8 +244,9 @@ class PreviewGaussianSplat(IO.ComfyNode): filename = f"preview_splat_{uuid.uuid4().hex}.{model_3d.format}" model_3d.save_to(os.path.join(folder_paths.get_temp_directory(), filename)) + viewport_state = viewport_state if isinstance(viewport_state, dict) else {} camera_info_input = kwargs.get("camera_info", None) - camera_info = camera_info_input if camera_info_input is not None else viewport_state['camera_info'] + camera_info = camera_info_input if camera_info_input is not None else viewport_state.get('camera_info') model_3d_info_input = kwargs.get("model_3d_info", None) model_3d_info = model_3d_info_input if model_3d_info_input is not None else viewport_state.get('model_3d_info', []) return IO.NodeOutput( @@ -303,8 +305,9 @@ class PreviewPointCloud(IO.ComfyNode): filename = f"preview_pointcloud_{uuid.uuid4().hex}.{model_3d.format}" model_3d.save_to(os.path.join(folder_paths.get_temp_directory(), filename)) + viewport_state = viewport_state if isinstance(viewport_state, dict) else {} camera_info_input = kwargs.get("camera_info", None) - camera_info = camera_info_input if camera_info_input is not None else viewport_state['camera_info'] + camera_info = camera_info_input if camera_info_input is not None else viewport_state.get('camera_info') model_3d_info_input = kwargs.get("model_3d_info", None) model_3d_info = model_3d_info_input if model_3d_info_input is not None else viewport_state.get('model_3d_info', []) return IO.NodeOutput( @@ -375,8 +378,9 @@ class Load3DAdvanced(IO.ComfyNode): file_3d = None if model_file and model_file != "none": file_3d = Types.File3D(folder_paths.get_annotated_filepath(model_file)) + viewport_state = viewport_state if isinstance(viewport_state, dict) else {} model_3d_info = viewport_state.get('model_3d_info', []) - return IO.NodeOutput(file_3d, model_3d_info, viewport_state['camera_info'], width, height) + return IO.NodeOutput(file_3d, model_3d_info, viewport_state.get('camera_info'), width, height) class Load3DExtension(ComfyExtension): diff --git a/comfy_extras/nodes_save_3d.py b/comfy_extras/nodes_save_3d.py index 7c524caa1..e9fd07326 100644 --- a/comfy_extras/nodes_save_3d.py +++ b/comfy_extras/nodes_save_3d.py @@ -418,8 +418,9 @@ def _save_file3d_to_output(model_3d: Types.File3D, filename_prefix: str) -> str: def execute_save_3d_advanced(model_3d, viewport_state, width, height, filename_prefix, kwargs) -> IO.NodeOutput: model_file = _save_file3d_to_output(model_3d, filename_prefix) + viewport_state = viewport_state if isinstance(viewport_state, dict) else {} camera_info_input = kwargs.get("camera_info", None) - camera_info = camera_info_input if camera_info_input is not None else viewport_state['camera_info'] + camera_info = camera_info_input if camera_info_input is not None else viewport_state.get('camera_info') model_3d_info_input = kwargs.get("model_3d_info", None) model_3d_info = model_3d_info_input if model_3d_info_input is not None else viewport_state.get('model_3d_info', []) return IO.NodeOutput(