Merge branch 'master' into deepme987/auto-register-node-replacements-json

2026-05-06 10:17:59 +08:00 · 2026-04-28 02:53:37 -07:00 · 2026-04-21 05:00:15 +05:30 · 2026-04-14 19:34:21 -07:00 · 2026-04-06 13:13:42 -07:00 · 2026-03-25 22:50:44 -07:00
28 changed files with 404 additions and 1738 deletions
--- a/2
+++ b/2
@ -1,2 +1,2 @@
 # Admins
-* @comfyanonymous @kosinkadink @guill @alexisrolland @rattus128
+* @comfyanonymous @kosinkadink @guill
--- a/app/node_replace_manager.py
+++ b/app/node_replace_manager.py
@ -1,5 +1,9 @@
 from __future__ import annotations

+import json
+import logging
+import os
+
 from aiohttp import web

 from typing import TYPE_CHECKING, TypedDict
@ -7,7 +11,6 @@ if TYPE_CHECKING:
    from comfy_api.latest._io_public import NodeReplace

 from comfy_execution.graph_utils import is_link
-import nodes

 class NodeStruct(TypedDict):
    inputs: dict[str, str | int | float | bool | tuple[str, int]]
@ -43,6 +46,7 @@ class NodeReplaceManager:
        return old_node_id in self._replacements

    def apply_replacements(self, prompt: dict[str, NodeStruct]):
+        import nodes
        connections: dict[str, list[tuple[str, str, int]]] = {}
        need_replacement: set[str] = set()
        for node_number, node_struct in prompt.items():
@ -94,6 +98,60 @@ class NodeReplaceManager:
                                previous_input = prompt[conn_node_number]["inputs"][conn_input_id]
                                previous_input[1] = new_output_idx

+    def load_from_json(self, module_dir: str, module_name: str, _node_replace_class=None):
+        """Load node_replacements.json from a custom node directory and register replacements.
+
+        Custom node authors can ship a node_replacements.json file in their repo root
+        to define node replacements declaratively. The file format matches the output
+        of NodeReplace.as_dict(), keyed by old_node_id.
+
+        Fail-open: all errors are logged and skipped so a malformed file never
+        prevents the custom node from loading.
+        """
+        replacements_path = os.path.join(module_dir, "node_replacements.json")
+        if not os.path.isfile(replacements_path):
+            return
+
+        try:
+            with open(replacements_path, "r", encoding="utf-8") as f:
+                data = json.load(f)
+
+            if not isinstance(data, dict):
+                logging.warning(f"node_replacements.json in {module_name} must be a JSON object, skipping.")
+                return
+
+            if _node_replace_class is None:
+                from comfy_api.latest._io import NodeReplace
+                _node_replace_class = NodeReplace
+
+            count = 0
+            for old_node_id, replacements in data.items():
+                if not isinstance(replacements, list):
+                    logging.warning(f"node_replacements.json in {module_name}: value for '{old_node_id}' must be a list, skipping.")
+                    continue
+                for entry in replacements:
+                    if not isinstance(entry, dict):
+                        continue
+                    new_node_id = entry.get("new_node_id", "")
+                    if not new_node_id:
+                        logging.warning(f"node_replacements.json in {module_name}: entry for '{old_node_id}' missing 'new_node_id', skipping.")
+                        continue
+                    self.register(_node_replace_class(
+                        new_node_id=new_node_id,
+                        old_node_id=entry.get("old_node_id", old_node_id),
+                        old_widget_ids=entry.get("old_widget_ids"),
+                        input_mapping=entry.get("input_mapping"),
+                        output_mapping=entry.get("output_mapping"),
+                    ))
+                    count += 1
+
+            if count > 0:
+                logging.info(f"Loaded {count} node replacement(s) from {module_name}/node_replacements.json")
+        except json.JSONDecodeError as e:
+            logging.warning(f"Failed to parse node_replacements.json in {module_name}: {e}")
+        except Exception as e:
+            logging.warning(f"Failed to load node_replacements.json from {module_name}: {e}")
+
    def as_dict(self):
        """Serialize all replacements to dict."""
        return {
--- a/comfy/latent_formats.py
+++ b/comfy/latent_formats.py
@ -224,7 +224,6 @@ class Flux2(LatentFormat):

        self.latent_rgb_factors_bias = [-0.0329, -0.0718, -0.0851]
        self.latent_rgb_factors_reshape = lambda t: t.reshape(t.shape[0], 32, 2, 2, t.shape[-2], t.shape[-1]).permute(0, 1, 4, 2, 5, 3).reshape(t.shape[0], 32, t.shape[-2] * 2, t.shape[-1] * 2)
-        self.taesd_decoder_name = "taef2_decoder"

    def process_in(self, latent):
        return latent
@ -784,10 +783,3 @@ class ZImagePixelSpace(ChromaRadiance):
    No VAE encoding/decoding — the model operates directly on RGB pixels.
    """
    pass
-
-class CogVideoX(LatentFormat):
-    latent_channels = 16
-    latent_dimensions = 3
-
-    def __init__(self):
-        self.scale_factor = 1.15258426
--- a/comfy/ldm/cogvideo/init.py
+++ b/comfy/ldm/cogvideo/init.py
--- a/comfy/ldm/cogvideo/model.py
+++ b/comfy/ldm/cogvideo/model.py
@ -1,573 +0,0 @@
-# CogVideoX 3D Transformer - ported to ComfyUI native ops
-# Architecture reference: diffusers CogVideoXTransformer3DModel
-# Style reference: comfy/ldm/wan/model.py
-
-import math
-import torch
-import torch.nn as nn
-import torch.nn.functional as F
-
-from comfy.ldm.modules.attention import optimized_attention
-import comfy.patcher_extension
-import comfy.ldm.common_dit
-
-
-def _get_1d_rotary_pos_embed(dim, pos, theta=10000.0):
-    """Returns (cos, sin) each with shape [seq_len, dim].
-
-    Frequencies are computed at dim//2 resolution then repeat_interleaved
-    to full dim, matching CogVideoX's interleaved (real, imag) pair format.
-    """
-    freqs = 1.0 / (theta ** (torch.arange(0, dim, 2, dtype=torch.float32, device=pos.device) / dim))
-    angles = torch.outer(pos.float(), freqs.float())
-    cos = angles.cos().repeat_interleave(2, dim=-1).float()
-    sin = angles.sin().repeat_interleave(2, dim=-1).float()
-    return (cos, sin)
-
-
-def apply_rotary_emb(x, freqs_cos_sin):
-    """Apply CogVideoX rotary embedding to query or key tensor.
-
-    x: [B, heads, seq_len, head_dim]
-    freqs_cos_sin: (cos, sin) each [seq_len, head_dim//2]
-
-    Uses interleaved pair rotation (same as diffusers CogVideoX/Flux).
-    head_dim is reshaped to (-1, 2) pairs, rotated, then flattened back.
-    """
-    cos, sin = freqs_cos_sin
-    cos = cos[None, None, :, :].to(x.device)
-    sin = sin[None, None, :, :].to(x.device)
-
-    # Interleaved pairs: [B, H, S, D] -> [B, H, S, D//2, 2] -> (real, imag)
-    x_real, x_imag = x.reshape(*x.shape[:-1], -1, 2).unbind(-1)
-    x_rotated = torch.stack([-x_imag, x_real], dim=-1).flatten(3)
-
-    return (x.float() * cos + x_rotated.float() * sin).to(x.dtype)
-
-
-def get_timestep_embedding(timesteps, dim, flip_sin_to_cos=True, downscale_freq_shift=0, scale=1, max_period=10000):
-    half = dim // 2
-    freqs = torch.exp(-math.log(max_period) * torch.arange(start=0, end=half, dtype=torch.float32, device=timesteps.device) / half)
-    args = timesteps[:, None].float() * freqs[None] * scale
-    embedding = torch.cat([torch.sin(args), torch.cos(args)], dim=-1)
-    if flip_sin_to_cos:
-        embedding = torch.cat([embedding[:, half:], embedding[:, :half]], dim=-1)
-    if dim % 2:
-        embedding = torch.cat([embedding, torch.zeros_like(embedding[:, :1])], dim=-1)
-    return embedding
-
-
-def get_3d_sincos_pos_embed(embed_dim, spatial_size, temporal_size, spatial_interpolation_scale=1.0, temporal_interpolation_scale=1.0, device=None):
-    if isinstance(spatial_size, int):
-        spatial_size = (spatial_size, spatial_size)
-
-    grid_w = torch.arange(spatial_size[0], dtype=torch.float32, device=device) / spatial_interpolation_scale
-    grid_h = torch.arange(spatial_size[1], dtype=torch.float32, device=device) / spatial_interpolation_scale
-    grid_t = torch.arange(temporal_size, dtype=torch.float32, device=device) / temporal_interpolation_scale
-
-    grid_t, grid_h, grid_w = torch.meshgrid(grid_t, grid_h, grid_w, indexing="ij")
-
-    embed_dim_spatial = 2 * (embed_dim // 3)
-    embed_dim_temporal = embed_dim // 3
-
-    pos_embed_spatial = _get_2d_sincos_pos_embed(embed_dim_spatial, grid_h, grid_w, device=device)
-    pos_embed_temporal = _get_1d_sincos_pos_embed(embed_dim_temporal, grid_t[:, 0, 0], device=device)
-
-    T, H, W = grid_t.shape
-    pos_embed_temporal = pos_embed_temporal.unsqueeze(1).unsqueeze(1).expand(-1, H, W, -1)
-    pos_embed = torch.cat([pos_embed_temporal, pos_embed_spatial], dim=-1)
-
-    return pos_embed
-
-
-def _get_2d_sincos_pos_embed(embed_dim, grid_h, grid_w, device=None):
-    T, H, W = grid_h.shape
-    half_dim = embed_dim // 2
-    pos_h = _get_1d_sincos_pos_embed(half_dim, grid_h.reshape(-1), device=device).reshape(T, H, W, half_dim)
-    pos_w = _get_1d_sincos_pos_embed(half_dim, grid_w.reshape(-1), device=device).reshape(T, H, W, half_dim)
-    return torch.cat([pos_h, pos_w], dim=-1)
-
-
-def _get_1d_sincos_pos_embed(embed_dim, pos, device=None):
-    half = embed_dim // 2
-    freqs = torch.exp(-math.log(10000.0) * torch.arange(start=0, end=half, dtype=torch.float32, device=device) / half)
-    args = pos.float().reshape(-1)[:, None] * freqs[None]
-    embedding = torch.cat([torch.cos(args), torch.sin(args)], dim=-1)
-    if embed_dim % 2:
-        embedding = torch.cat([embedding, torch.zeros_like(embedding[:, :1])], dim=-1)
-    return embedding
-
-
-
-class CogVideoXPatchEmbed(nn.Module):
-    def __init__(self, patch_size=2, patch_size_t=None, in_channels=16, dim=1920,
-                 text_dim=4096, bias=True, sample_width=90, sample_height=60,
-                 sample_frames=49, temporal_compression_ratio=4,
-                 max_text_seq_length=226, spatial_interpolation_scale=1.875,
-                 temporal_interpolation_scale=1.0, use_positional_embeddings=True,
-                 use_learned_positional_embeddings=True,
-                 device=None, dtype=None, operations=None):
-        super().__init__()
-        self.patch_size = patch_size
-        self.patch_size_t = patch_size_t
-        self.dim = dim
-        self.sample_height = sample_height
-        self.sample_width = sample_width
-        self.sample_frames = sample_frames
-        self.temporal_compression_ratio = temporal_compression_ratio
-        self.max_text_seq_length = max_text_seq_length
-        self.spatial_interpolation_scale = spatial_interpolation_scale
-        self.temporal_interpolation_scale = temporal_interpolation_scale
-        self.use_positional_embeddings = use_positional_embeddings
-        self.use_learned_positional_embeddings = use_learned_positional_embeddings
-
-        if patch_size_t is None:
-            self.proj = operations.Conv2d(in_channels, dim, kernel_size=patch_size, stride=patch_size, bias=bias, device=device, dtype=dtype)
-        else:
-            self.proj = operations.Linear(in_channels * patch_size * patch_size * patch_size_t, dim, device=device, dtype=dtype)
-
-        self.text_proj = operations.Linear(text_dim, dim, device=device, dtype=dtype)
-
-        if use_positional_embeddings or use_learned_positional_embeddings:
-            persistent = use_learned_positional_embeddings
-            pos_embedding = self._get_positional_embeddings(sample_height, sample_width, sample_frames)
-            self.register_buffer("pos_embedding", pos_embedding, persistent=persistent)
-
-    def _get_positional_embeddings(self, sample_height, sample_width, sample_frames, device=None):
-        post_patch_height = sample_height // self.patch_size
-        post_patch_width = sample_width // self.patch_size
-        post_time_compression_frames = (sample_frames - 1) // self.temporal_compression_ratio + 1
-        if self.patch_size_t is not None:
-            post_time_compression_frames = post_time_compression_frames // self.patch_size_t
-        num_patches = post_patch_height * post_patch_width * post_time_compression_frames
-
-        pos_embedding = get_3d_sincos_pos_embed(
-            self.dim,
-            (post_patch_width, post_patch_height),
-            post_time_compression_frames,
-            self.spatial_interpolation_scale,
-            self.temporal_interpolation_scale,
-            device=device,
-        )
-        pos_embedding = pos_embedding.reshape(-1, self.dim)
-        joint_pos_embedding = pos_embedding.new_zeros(
-            1, self.max_text_seq_length + num_patches, self.dim, requires_grad=False
-        )
-        joint_pos_embedding.data[:, self.max_text_seq_length:].copy_(pos_embedding)
-        return joint_pos_embedding
-
-    def forward(self, text_embeds, image_embeds):
-        input_dtype = text_embeds.dtype
-        text_embeds = self.text_proj(text_embeds.to(self.text_proj.weight.dtype)).to(input_dtype)
-        batch_size, num_frames, channels, height, width = image_embeds.shape
-
-        proj_dtype = self.proj.weight.dtype
-        if self.patch_size_t is None:
-            image_embeds = image_embeds.reshape(-1, channels, height, width)
-            image_embeds = self.proj(image_embeds.to(proj_dtype)).to(input_dtype)
-            image_embeds = image_embeds.view(batch_size, num_frames, *image_embeds.shape[1:])
-            image_embeds = image_embeds.flatten(3).transpose(2, 3)
-            image_embeds = image_embeds.flatten(1, 2)
-        else:
-            p = self.patch_size
-            p_t = self.patch_size_t
-            image_embeds = image_embeds.permute(0, 1, 3, 4, 2)
-            image_embeds = image_embeds.reshape(
-                batch_size, num_frames // p_t, p_t, height // p, p, width // p, p, channels
-            )
-            image_embeds = image_embeds.permute(0, 1, 3, 5, 7, 2, 4, 6).flatten(4, 7).flatten(1, 3)
-            image_embeds = self.proj(image_embeds.to(proj_dtype)).to(input_dtype)
-
-        embeds = torch.cat([text_embeds, image_embeds], dim=1).contiguous()
-
-        if self.use_positional_embeddings or self.use_learned_positional_embeddings:
-            text_seq_length = text_embeds.shape[1]
-            num_image_patches = image_embeds.shape[1]
-
-            if self.use_learned_positional_embeddings:
-                image_pos = self.pos_embedding[
-                    :, self.max_text_seq_length:self.max_text_seq_length + num_image_patches
-                ].to(device=embeds.device, dtype=embeds.dtype)
-            else:
-                image_pos = get_3d_sincos_pos_embed(
-                    self.dim,
-                    (width // self.patch_size, height // self.patch_size),
-                    num_image_patches // ((height // self.patch_size) * (width // self.patch_size)),
-                    self.spatial_interpolation_scale,
-                    self.temporal_interpolation_scale,
-                    device=embeds.device,
-                ).reshape(1, num_image_patches, self.dim).to(dtype=embeds.dtype)
-
-            # Build joint: zeros for text + sincos for image
-            joint_pos = torch.zeros(1, text_seq_length + num_image_patches, self.dim, device=embeds.device, dtype=embeds.dtype)
-            joint_pos[:, text_seq_length:] = image_pos
-            embeds = embeds + joint_pos
-
-        return embeds
-
-
-class CogVideoXLayerNormZero(nn.Module):
-    def __init__(self, time_dim, dim, elementwise_affine=True, eps=1e-5, bias=True,
-                 device=None, dtype=None, operations=None):
-        super().__init__()
-        self.silu = nn.SiLU()
-        self.linear = operations.Linear(time_dim, 6 * dim, bias=bias, device=device, dtype=dtype)
-        self.norm = operations.LayerNorm(dim, eps=eps, elementwise_affine=elementwise_affine, device=device, dtype=dtype)
-
-    def forward(self, hidden_states, encoder_hidden_states, temb):
-        shift, scale, gate, enc_shift, enc_scale, enc_gate = self.linear(self.silu(temb)).chunk(6, dim=1)
-        hidden_states = self.norm(hidden_states) * (1 + scale)[:, None, :] + shift[:, None, :]
-        encoder_hidden_states = self.norm(encoder_hidden_states) * (1 + enc_scale)[:, None, :] + enc_shift[:, None, :]
-        return hidden_states, encoder_hidden_states, gate[:, None, :], enc_gate[:, None, :]
-
-
-class CogVideoXAdaLayerNorm(nn.Module):
-    def __init__(self, time_dim, dim, elementwise_affine=True, eps=1e-5,
-                 device=None, dtype=None, operations=None):
-        super().__init__()
-        self.silu = nn.SiLU()
-        self.linear = operations.Linear(time_dim, 2 * dim, device=device, dtype=dtype)
-        self.norm = operations.LayerNorm(dim, eps=eps, elementwise_affine=elementwise_affine, device=device, dtype=dtype)
-
-    def forward(self, x, temb):
-        temb = self.linear(self.silu(temb))
-        shift, scale = temb.chunk(2, dim=1)
-        x = self.norm(x) * (1 + scale)[:, None, :] + shift[:, None, :]
-        return x
-
-
-class CogVideoXBlock(nn.Module):
-    def __init__(self, dim, num_heads, head_dim, time_dim,
-                 eps=1e-5, ff_inner_dim=None, ff_bias=True,
-                 device=None, dtype=None, operations=None):
-        super().__init__()
-        self.dim = dim
-        self.num_heads = num_heads
-        self.head_dim = head_dim
-
-        self.norm1 = CogVideoXLayerNormZero(time_dim, dim, eps=eps, device=device, dtype=dtype, operations=operations)
-
-        # Self-attention (joint text + latent)
-        self.q = operations.Linear(dim, dim, bias=True, device=device, dtype=dtype)
-        self.k = operations.Linear(dim, dim, bias=True, device=device, dtype=dtype)
-        self.v = operations.Linear(dim, dim, bias=True, device=device, dtype=dtype)
-        self.norm_q = operations.LayerNorm(head_dim, eps=1e-6, elementwise_affine=True, device=device, dtype=dtype)
-        self.norm_k = operations.LayerNorm(head_dim, eps=1e-6, elementwise_affine=True, device=device, dtype=dtype)
-        self.attn_out = operations.Linear(dim, dim, bias=True, device=device, dtype=dtype)
-
-        self.norm2 = CogVideoXLayerNormZero(time_dim, dim, eps=eps, device=device, dtype=dtype, operations=operations)
-
-        # Feed-forward (GELU approximate)
-        inner_dim = ff_inner_dim or dim * 4
-        self.ff_proj = operations.Linear(dim, inner_dim, bias=ff_bias, device=device, dtype=dtype)
-        self.ff_out = operations.Linear(inner_dim, dim, bias=ff_bias, device=device, dtype=dtype)
-
-    def forward(self, hidden_states, encoder_hidden_states, temb, image_rotary_emb=None, transformer_options=None):
-        if transformer_options is None:
-            transformer_options = {}
-        text_seq_length = encoder_hidden_states.size(1)
-
-        # Norm & modulate
-        norm_hidden, norm_encoder, gate_msa, enc_gate_msa = self.norm1(hidden_states, encoder_hidden_states, temb)
-
-        # Joint self-attention
-        qkv_input = torch.cat([norm_encoder, norm_hidden], dim=1)
-        b, s, _ = qkv_input.shape
-        n, d = self.num_heads, self.head_dim
-
-        q = self.q(qkv_input).view(b, s, n, d)
-        k = self.k(qkv_input).view(b, s, n, d)
-        v = self.v(qkv_input)
-
-        q = self.norm_q(q).view(b, s, n, d)
-        k = self.norm_k(k).view(b, s, n, d)
-
-        # Apply rotary embeddings to image tokens only (diffusers format: [B, heads, seq, head_dim])
-        if image_rotary_emb is not None:
-            q_img = q[:, text_seq_length:].transpose(1, 2)  # [B, heads, img_seq, head_dim]
-            k_img = k[:, text_seq_length:].transpose(1, 2)
-            q_img = apply_rotary_emb(q_img, image_rotary_emb)
-            k_img = apply_rotary_emb(k_img, image_rotary_emb)
-            q = torch.cat([q[:, :text_seq_length], q_img.transpose(1, 2)], dim=1)
-            k = torch.cat([k[:, :text_seq_length], k_img.transpose(1, 2)], dim=1)
-
-        attn_out = optimized_attention(
-            q.reshape(b, s, n * d),
-            k.reshape(b, s, n * d),
-            v,
-            heads=self.num_heads,
-            transformer_options=transformer_options,
-        )
-
-        attn_out = self.attn_out(attn_out)
-
-        attn_encoder, attn_hidden = attn_out.split([text_seq_length, s - text_seq_length], dim=1)
-
-        hidden_states = hidden_states + gate_msa * attn_hidden
-        encoder_hidden_states = encoder_hidden_states + enc_gate_msa * attn_encoder
-
-        # Norm & modulate for FF
-        norm_hidden, norm_encoder, gate_ff, enc_gate_ff = self.norm2(hidden_states, encoder_hidden_states, temb)
-
-        # Feed-forward (GELU on concatenated text + latent)
-        ff_input = torch.cat([norm_encoder, norm_hidden], dim=1)
-        ff_output = self.ff_out(F.gelu(self.ff_proj(ff_input), approximate="tanh"))
-
-        hidden_states = hidden_states + gate_ff * ff_output[:, text_seq_length:]
-        encoder_hidden_states = encoder_hidden_states + enc_gate_ff * ff_output[:, :text_seq_length]
-
-        return hidden_states, encoder_hidden_states
-
-
-class CogVideoXTransformer3DModel(nn.Module):
-    def __init__(self,
-                 num_attention_heads=30,
-                 attention_head_dim=64,
-                 in_channels=16,
-                 out_channels=16,
-                 flip_sin_to_cos=True,
-                 freq_shift=0,
-                 time_embed_dim=512,
-                 ofs_embed_dim=None,
-                 text_embed_dim=4096,
-                 num_layers=30,
-                 dropout=0.0,
-                 attention_bias=True,
-                 sample_width=90,
-                 sample_height=60,
-                 sample_frames=49,
-                 patch_size=2,
-                 patch_size_t=None,
-                 temporal_compression_ratio=4,
-                 max_text_seq_length=226,
-                 spatial_interpolation_scale=1.875,
-                 temporal_interpolation_scale=1.0,
-                 use_rotary_positional_embeddings=False,
-                 use_learned_positional_embeddings=False,
-                 patch_bias=True,
-                 image_model=None,
-                 device=None,
-                 dtype=None,
-                 operations=None,
-                 ):
-        super().__init__()
-        self.dtype = dtype
-        dim = num_attention_heads * attention_head_dim
-        self.dim = dim
-        self.num_attention_heads = num_attention_heads
-        self.attention_head_dim = attention_head_dim
-        self.in_channels = in_channels
-        self.out_channels = out_channels
-        self.patch_size = patch_size
-        self.patch_size_t = patch_size_t
-        self.max_text_seq_length = max_text_seq_length
-        self.use_rotary_positional_embeddings = use_rotary_positional_embeddings
-
-        # 1. Patch embedding
-        self.patch_embed = CogVideoXPatchEmbed(
-            patch_size=patch_size,
-            patch_size_t=patch_size_t,
-            in_channels=in_channels,
-            dim=dim,
-            text_dim=text_embed_dim,
-            bias=patch_bias,
-            sample_width=sample_width,
-            sample_height=sample_height,
-            sample_frames=sample_frames,
-            temporal_compression_ratio=temporal_compression_ratio,
-            max_text_seq_length=max_text_seq_length,
-            spatial_interpolation_scale=spatial_interpolation_scale,
-            temporal_interpolation_scale=temporal_interpolation_scale,
-            use_positional_embeddings=not use_rotary_positional_embeddings,
-            use_learned_positional_embeddings=use_learned_positional_embeddings,
-            device=device, dtype=torch.float32, operations=operations,
-        )
-
-        # 2. Time embedding
-        self.time_proj_dim = dim
-        self.time_proj_flip = flip_sin_to_cos
-        self.time_proj_shift = freq_shift
-        self.time_embedding_linear_1 = operations.Linear(dim, time_embed_dim, device=device, dtype=dtype)
-        self.time_embedding_act = nn.SiLU()
-        self.time_embedding_linear_2 = operations.Linear(time_embed_dim, time_embed_dim, device=device, dtype=dtype)
-
-        # Optional OFS embedding (CogVideoX 1.5 I2V)
-        self.ofs_proj_dim = ofs_embed_dim
-        if ofs_embed_dim:
-            self.ofs_embedding_linear_1 = operations.Linear(ofs_embed_dim, ofs_embed_dim, device=device, dtype=dtype)
-            self.ofs_embedding_act = nn.SiLU()
-            self.ofs_embedding_linear_2 = operations.Linear(ofs_embed_dim, ofs_embed_dim, device=device, dtype=dtype)
-        else:
-            self.ofs_embedding_linear_1 = None
-
-        # 3. Transformer blocks
-        self.blocks = nn.ModuleList([
-            CogVideoXBlock(
-                dim=dim,
-                num_heads=num_attention_heads,
-                head_dim=attention_head_dim,
-                time_dim=time_embed_dim,
-                eps=1e-5,
-                device=device, dtype=dtype, operations=operations,
-            )
-            for _ in range(num_layers)
-        ])
-
-        self.norm_final = operations.LayerNorm(dim, eps=1e-5, elementwise_affine=True, device=device, dtype=dtype)
-
-        # 4. Output
-        self.norm_out = CogVideoXAdaLayerNorm(
-            time_dim=time_embed_dim, dim=dim, eps=1e-5,
-            device=device, dtype=dtype, operations=operations,
-        )
-
-        if patch_size_t is None:
-            output_dim = patch_size * patch_size * out_channels
-        else:
-            output_dim = patch_size * patch_size * patch_size_t * out_channels
-
-        self.proj_out = operations.Linear(dim, output_dim, device=device, dtype=dtype)
-
-        self.spatial_interpolation_scale = spatial_interpolation_scale
-        self.temporal_interpolation_scale = temporal_interpolation_scale
-        self.temporal_compression_ratio = temporal_compression_ratio
-
-    def forward(self, x, timestep, context, ofs=None, transformer_options=None, **kwargs):
-        if transformer_options is None:
-            transformer_options = {}
-        return comfy.patcher_extension.WrapperExecutor.new_class_executor(
-            self._forward,
-            self,
-            comfy.patcher_extension.get_all_wrappers(comfy.patcher_extension.WrappersMP.DIFFUSION_MODEL, transformer_options)
-        ).execute(x, timestep, context, ofs, transformer_options, **kwargs)
-
-    def _forward(self, x, timestep, context, ofs=None, transformer_options=None, **kwargs):
-        if transformer_options is None:
-            transformer_options = {}
-        # ComfyUI passes [B, C, T, H, W]
-        batch_size, channels, t, h, w = x.shape
-
-        # Pad to patch size (temporal + spatial), same pattern as WAN
-        p_t = self.patch_size_t if self.patch_size_t is not None else 1
-        x = comfy.ldm.common_dit.pad_to_patch_size(x, (p_t, self.patch_size, self.patch_size))
-
-        # CogVideoX expects [B, T, C, H, W]
-        x = x.permute(0, 2, 1, 3, 4)
-        batch_size, num_frames, channels, height, width = x.shape
-
-        # Time embedding
-        t_emb = get_timestep_embedding(timestep, self.time_proj_dim, self.time_proj_flip, self.time_proj_shift)
-        t_emb = t_emb.to(dtype=x.dtype)
-        emb = self.time_embedding_linear_2(self.time_embedding_act(self.time_embedding_linear_1(t_emb)))
-
-        if self.ofs_embedding_linear_1 is not None and ofs is not None:
-            ofs_emb = get_timestep_embedding(ofs, self.ofs_proj_dim, self.time_proj_flip, self.time_proj_shift)
-            ofs_emb = ofs_emb.to(dtype=x.dtype)
-            ofs_emb = self.ofs_embedding_linear_2(self.ofs_embedding_act(self.ofs_embedding_linear_1(ofs_emb)))
-            emb = emb + ofs_emb
-
-        # Patch embedding
-        hidden_states = self.patch_embed(context, x)
-
-        text_seq_length = context.shape[1]
-        encoder_hidden_states = hidden_states[:, :text_seq_length]
-        hidden_states = hidden_states[:, text_seq_length:]
-
-        # Rotary embeddings (if used)
-        image_rotary_emb = None
-        if self.use_rotary_positional_embeddings:
-            post_patch_height = height // self.patch_size
-            post_patch_width = width // self.patch_size
-            if self.patch_size_t is None:
-                post_time = num_frames
-            else:
-                post_time = num_frames // self.patch_size_t
-            image_rotary_emb = self._get_rotary_emb(post_patch_height, post_patch_width, post_time, device=x.device)
-
-        # Transformer blocks
-        for i, block in enumerate(self.blocks):
-            hidden_states, encoder_hidden_states = block(
-                hidden_states=hidden_states,
-                encoder_hidden_states=encoder_hidden_states,
-                temb=emb,
-                image_rotary_emb=image_rotary_emb,
-                transformer_options=transformer_options,
-            )
-
-        hidden_states = self.norm_final(hidden_states)
-
-        # Output projection
-        hidden_states = self.norm_out(hidden_states, temb=emb)
-        hidden_states = self.proj_out(hidden_states)
-
-        # Unpatchify
-        p = self.patch_size
-        p_t = self.patch_size_t
-
-        if p_t is None:
-            output = hidden_states.reshape(batch_size, num_frames, height // p, width // p, -1, p, p)
-            output = output.permute(0, 1, 4, 2, 5, 3, 6).flatten(5, 6).flatten(3, 4)
-        else:
-            output = hidden_states.reshape(
-                batch_size, (num_frames + p_t - 1) // p_t, height // p, width // p, -1, p_t, p, p
-            )
-            output = output.permute(0, 1, 5, 4, 2, 6, 3, 7).flatten(6, 7).flatten(4, 5).flatten(1, 2)
-
-        # Back to ComfyUI format [B, C, T, H, W] and crop padding
-        output = output.permute(0, 2, 1, 3, 4)[:, :, :t, :h, :w]
-        return output
-
-    def _get_rotary_emb(self, h, w, t, device):
-        """Compute CogVideoX 3D rotary positional embeddings.
-
-        For CogVideoX 1.5 (patch_size_t != None): uses "slice" mode — grid positions
-        are integer arange computed at max_size, then sliced to actual size.
-        For CogVideoX 1.0 (patch_size_t == None): uses "linspace" mode with crop coords
-        scaled by spatial_interpolation_scale.
-        """
-        d = self.attention_head_dim
-        dim_t = d // 4
-        dim_h = d // 8 * 3
-        dim_w = d // 8 * 3
-
-        if self.patch_size_t is not None:
-            # CogVideoX 1.5: "slice" mode — positions are simple integer indices
-            # Compute at max(sample_size, actual_size) then slice to actual
-            base_h = self.patch_embed.sample_height // self.patch_size
-            base_w = self.patch_embed.sample_width // self.patch_size
-            max_h = max(base_h, h)
-            max_w = max(base_w, w)
-
-            grid_h = torch.arange(max_h, device=device, dtype=torch.float32)
-            grid_w = torch.arange(max_w, device=device, dtype=torch.float32)
-            grid_t = torch.arange(t, device=device, dtype=torch.float32)
-        else:
-            # CogVideoX 1.0: "linspace" mode with interpolation scale
-            grid_h = torch.linspace(0, h - 1, h, device=device, dtype=torch.float32) * self.spatial_interpolation_scale
-            grid_w = torch.linspace(0, w - 1, w, device=device, dtype=torch.float32) * self.spatial_interpolation_scale
-            grid_t = torch.arange(t, device=device, dtype=torch.float32)
-
-        freqs_t = _get_1d_rotary_pos_embed(dim_t, grid_t)
-        freqs_h = _get_1d_rotary_pos_embed(dim_h, grid_h)
-        freqs_w = _get_1d_rotary_pos_embed(dim_w, grid_w)
-
-        t_cos, t_sin = freqs_t
-        h_cos, h_sin = freqs_h
-        w_cos, w_sin = freqs_w
-
-        # Slice to actual size (for "slice" mode where grids may be larger)
-        t_cos, t_sin = t_cos[:t], t_sin[:t]
-        h_cos, h_sin = h_cos[:h], h_sin[:h]
-        w_cos, w_sin = w_cos[:w], w_sin[:w]
-
-        # Broadcast and concatenate into [T*H*W, head_dim]
-        t_cos = t_cos[:, None, None, :].expand(-1, h, w, -1)
-        t_sin = t_sin[:, None, None, :].expand(-1, h, w, -1)
-        h_cos = h_cos[None, :, None, :].expand(t, -1, w, -1)
-        h_sin = h_sin[None, :, None, :].expand(t, -1, w, -1)
-        w_cos = w_cos[None, None, :, :].expand(t, h, -1, -1)
-        w_sin = w_sin[None, None, :, :].expand(t, h, -1, -1)
-
-        cos = torch.cat([t_cos, h_cos, w_cos], dim=-1).reshape(t * h * w, -1)
-        sin = torch.cat([t_sin, h_sin, w_sin], dim=-1).reshape(t * h * w, -1)
-        return (cos, sin)
--- a/comfy/ldm/cogvideo/vae.py
+++ b/comfy/ldm/cogvideo/vae.py
@ -1,566 +0,0 @@
-# CogVideoX VAE - ported to ComfyUI native ops
-# Architecture reference: diffusers AutoencoderKLCogVideoX
-# Style reference: comfy/ldm/wan/vae.py
-
-import numpy as np
-
-import torch
-import torch.nn as nn
-import torch.nn.functional as F
-
-import comfy.ops
-ops = comfy.ops.disable_weight_init
-
-
-class CausalConv3d(nn.Module):
-    """Causal 3D convolution with temporal padding.
-
-    Uses comfy.ops.Conv3d with autopad='causal_zero' fast path: when input has
-    a single temporal frame and no cache, the 3D conv weight is sliced to act
-    as a 2D conv, avoiding computation on zero-padded temporal dimensions.
-    """
-    def __init__(self, in_channels, out_channels, kernel_size, stride=1, dilation=1, pad_mode="constant"):
-        super().__init__()
-        if isinstance(kernel_size, int):
-            kernel_size = (kernel_size,) * 3
-
-        time_kernel, height_kernel, width_kernel = kernel_size
-        self.time_kernel_size = time_kernel
-        self.pad_mode = pad_mode
-
-        height_pad = (height_kernel - 1) // 2
-        width_pad = (width_kernel - 1) // 2
-        self.time_causal_padding = (width_pad, width_pad, height_pad, height_pad, time_kernel - 1, 0)
-
-        stride = stride if isinstance(stride, tuple) else (stride, 1, 1)
-        dilation = (dilation, 1, 1)
-        self.conv = ops.Conv3d(
-            in_channels, out_channels, kernel_size,
-            stride=stride, dilation=dilation,
-            padding=(0, height_pad, width_pad),
-        )
-
-    def forward(self, x, conv_cache=None):
-        if self.pad_mode == "replicate":
-            x = F.pad(x, self.time_causal_padding, mode="replicate")
-            conv_cache = None
-        else:
-            kernel_t = self.time_kernel_size
-            if kernel_t > 1:
-                if conv_cache is None and x.shape[2] == 1:
-                    # Fast path: single frame, no cache. All temporal padding
-                    # frames are copies of the input (replicate-style), so the
-                    # 3D conv reduces to a 2D conv with summed temporal kernel.
-                    w = comfy.ops.cast_to_input(self.conv.weight, x)
-                    b = comfy.ops.cast_to_input(self.conv.bias, x) if self.conv.bias is not None else None
-                    w2d = w.sum(dim=2, keepdim=True)
-                    out = F.conv3d(x, w2d, b,
-                                   self.conv.stride, self.conv.padding,
-                                   self.conv.dilation, self.conv.groups)
-                    return out, None
-                cached = [conv_cache] if conv_cache is not None else [x[:, :, :1]] * (kernel_t - 1)
-                x = torch.cat(cached + [x], dim=2)
-            conv_cache = x[:, :, -self.time_kernel_size + 1:].clone() if self.time_kernel_size > 1 else None
-
-        out = self.conv(x)
-        return out, conv_cache
-
-
-def _interpolate_zq(zq, target_size):
-    """Interpolate latent z to target (T, H, W), matching CogVideoX's first-frame-special handling."""
-    t = target_size[0]
-    if t > 1 and t % 2 == 1:
-        z_first = F.interpolate(zq[:, :, :1], size=(1, target_size[1], target_size[2]))
-        z_rest = F.interpolate(zq[:, :, 1:], size=(t - 1, target_size[1], target_size[2]))
-        return torch.cat([z_first, z_rest], dim=2)
-    return F.interpolate(zq, size=target_size)
-
-
-class SpatialNorm3D(nn.Module):
-    """Spatially conditioned normalization."""
-    def __init__(self, f_channels, zq_channels, groups=32):
-        super().__init__()
-        self.norm_layer = ops.GroupNorm(num_channels=f_channels, num_groups=groups, eps=1e-6, affine=True)
-        self.conv_y = CausalConv3d(zq_channels, f_channels, kernel_size=1, stride=1)
-        self.conv_b = CausalConv3d(zq_channels, f_channels, kernel_size=1, stride=1)
-
-    def forward(self, f, zq, conv_cache=None):
-        new_cache = {}
-        conv_cache = conv_cache or {}
-
-        if zq.shape[-3:] != f.shape[-3:]:
-            zq = _interpolate_zq(zq, f.shape[-3:])
-
-        conv_y, new_cache["conv_y"] = self.conv_y(zq, conv_cache=conv_cache.get("conv_y"))
-        conv_b, new_cache["conv_b"] = self.conv_b(zq, conv_cache=conv_cache.get("conv_b"))
-
-        return self.norm_layer(f) * conv_y + conv_b, new_cache
-
-
-class ResnetBlock3D(nn.Module):
-    """3D ResNet block with optional spatial norm."""
-    def __init__(self, in_channels, out_channels=None, temb_channels=512, groups=32,
-                 eps=1e-6, act_fn="silu", spatial_norm_dim=None, pad_mode="first"):
-        super().__init__()
-        out_channels = out_channels or in_channels
-        self.in_channels = in_channels
-        self.out_channels = out_channels
-        self.spatial_norm_dim = spatial_norm_dim
-
-        if act_fn == "silu":
-            self.nonlinearity = nn.SiLU()
-        elif act_fn == "swish":
-            self.nonlinearity = nn.SiLU()
-        else:
-            self.nonlinearity = nn.SiLU()
-
-        if spatial_norm_dim is None:
-            self.norm1 = ops.GroupNorm(num_channels=in_channels, num_groups=groups, eps=eps)
-            self.norm2 = ops.GroupNorm(num_channels=out_channels, num_groups=groups, eps=eps)
-        else:
-            self.norm1 = SpatialNorm3D(in_channels, spatial_norm_dim, groups=groups)
-            self.norm2 = SpatialNorm3D(out_channels, spatial_norm_dim, groups=groups)
-
-        self.conv1 = CausalConv3d(in_channels, out_channels, kernel_size=3, pad_mode=pad_mode)
-
-        if temb_channels > 0:
-            self.temb_proj = ops.Linear(temb_channels, out_channels)
-
-        self.conv2 = CausalConv3d(out_channels, out_channels, kernel_size=3, pad_mode=pad_mode)
-
-        if in_channels != out_channels:
-            self.conv_shortcut = ops.Conv3d(in_channels, out_channels, kernel_size=1, stride=1, padding=0)
-        else:
-            self.conv_shortcut = None
-
-    def forward(self, x, temb=None, zq=None, conv_cache=None):
-        new_cache = {}
-        conv_cache = conv_cache or {}
-        residual = x
-
-        if zq is not None:
-            x, new_cache["norm1"] = self.norm1(x, zq, conv_cache=conv_cache.get("norm1"))
-        else:
-            x = self.norm1(x)
-
-        x = self.nonlinearity(x)
-        x, new_cache["conv1"] = self.conv1(x, conv_cache=conv_cache.get("conv1"))
-
-        if temb is not None and hasattr(self, "temb_proj"):
-            x = x + self.temb_proj(self.nonlinearity(temb))[:, :, None, None, None]
-
-        if zq is not None:
-            x, new_cache["norm2"] = self.norm2(x, zq, conv_cache=conv_cache.get("norm2"))
-        else:
-            x = self.norm2(x)
-
-        x = self.nonlinearity(x)
-        x, new_cache["conv2"] = self.conv2(x, conv_cache=conv_cache.get("conv2"))
-
-        if self.conv_shortcut is not None:
-            residual = self.conv_shortcut(residual)
-
-        return x + residual, new_cache
-
-
-class Downsample3D(nn.Module):
-    """3D downsampling with optional temporal compression."""
-    def __init__(self, in_channels, out_channels, kernel_size=3, stride=2, padding=0, compress_time=False):
-        super().__init__()
-        self.conv = ops.Conv2d(in_channels, out_channels, kernel_size=kernel_size, stride=stride, padding=padding)
-        self.compress_time = compress_time
-
-    def forward(self, x):
-        if self.compress_time:
-            b, c, t, h, w = x.shape
-            x = x.permute(0, 3, 4, 1, 2).reshape(b * h * w, c, t)
-            if t % 2 == 1:
-                x_first, x_rest = x[..., 0], x[..., 1:]
-                if x_rest.shape[-1] > 0:
-                    x_rest = F.avg_pool1d(x_rest, kernel_size=2, stride=2)
-                x = torch.cat([x_first[..., None], x_rest], dim=-1)
-                x = x.reshape(b, h, w, c, x.shape[-1]).permute(0, 3, 4, 1, 2)
-            else:
-                x = F.avg_pool1d(x, kernel_size=2, stride=2)
-                x = x.reshape(b, h, w, c, x.shape[-1]).permute(0, 3, 4, 1, 2)
-
-        pad = (0, 1, 0, 1)
-        x = F.pad(x, pad, mode="constant", value=0)
-        b, c, t, h, w = x.shape
-        x = x.permute(0, 2, 1, 3, 4).reshape(b * t, c, h, w)
-        x = self.conv(x)
-        x = x.reshape(b, t, x.shape[1], x.shape[2], x.shape[3]).permute(0, 2, 1, 3, 4)
-        return x
-
-
-class Upsample3D(nn.Module):
-    """3D upsampling with optional temporal decompression."""
-    def __init__(self, in_channels, out_channels, kernel_size=3, stride=1, padding=1, compress_time=False):
-        super().__init__()
-        self.conv = ops.Conv2d(in_channels, out_channels, kernel_size=kernel_size, stride=stride, padding=padding)
-        self.compress_time = compress_time
-
-    def forward(self, x):
-        if self.compress_time:
-            if x.shape[2] > 1 and x.shape[2] % 2 == 1:
-                x_first, x_rest = x[:, :, 0], x[:, :, 1:]
-                x_first = F.interpolate(x_first, scale_factor=2.0)
-                x_rest = F.interpolate(x_rest, scale_factor=2.0)
-                x = torch.cat([x_first[:, :, None, :, :], x_rest], dim=2)
-            elif x.shape[2] > 1:
-                x = F.interpolate(x, scale_factor=2.0)
-            else:
-                x = x.squeeze(2)
-                x = F.interpolate(x, scale_factor=2.0)
-                x = x[:, :, None, :, :]
-        else:
-            b, c, t, h, w = x.shape
-            x = x.permute(0, 2, 1, 3, 4).reshape(b * t, c, h, w)
-            x = F.interpolate(x, scale_factor=2.0)
-            x = x.reshape(b, t, c, *x.shape[2:]).permute(0, 2, 1, 3, 4)
-
-        b, c, t, h, w = x.shape
-        x = x.permute(0, 2, 1, 3, 4).reshape(b * t, c, h, w)
-        x = self.conv(x)
-        x = x.reshape(b, t, *x.shape[1:]).permute(0, 2, 1, 3, 4)
-        return x
-
-
-class DownBlock3D(nn.Module):
-    def __init__(self, in_channels, out_channels, temb_channels=0, num_layers=1,
-                 eps=1e-6, act_fn="silu", groups=32, add_downsample=True,
-                 compress_time=False, pad_mode="first"):
-        super().__init__()
-        self.resnets = nn.ModuleList([
-            ResnetBlock3D(
-                in_channels=in_channels if i == 0 else out_channels,
-                out_channels=out_channels,
-                temb_channels=temb_channels,
-                groups=groups, eps=eps, act_fn=act_fn, pad_mode=pad_mode,
-            )
-            for i in range(num_layers)
-        ])
-        self.downsamplers = nn.ModuleList([Downsample3D(out_channels, out_channels, compress_time=compress_time)]) if add_downsample else None
-
-    def forward(self, x, temb=None, zq=None, conv_cache=None):
-        new_cache = {}
-        conv_cache = conv_cache or {}
-        for i, resnet in enumerate(self.resnets):
-            x, new_cache[f"resnet_{i}"] = resnet(x, temb, zq, conv_cache=conv_cache.get(f"resnet_{i}"))
-        if self.downsamplers is not None:
-            for ds in self.downsamplers:
-                x = ds(x)
-        return x, new_cache
-
-
-class MidBlock3D(nn.Module):
-    def __init__(self, in_channels, temb_channels=0, num_layers=1,
-                 eps=1e-6, act_fn="silu", groups=32, spatial_norm_dim=None, pad_mode="first"):
-        super().__init__()
-        self.resnets = nn.ModuleList([
-            ResnetBlock3D(
-                in_channels=in_channels, out_channels=in_channels,
-                temb_channels=temb_channels, groups=groups, eps=eps,
-                act_fn=act_fn, spatial_norm_dim=spatial_norm_dim, pad_mode=pad_mode,
-            )
-            for _ in range(num_layers)
-        ])
-
-    def forward(self, x, temb=None, zq=None, conv_cache=None):
-        new_cache = {}
-        conv_cache = conv_cache or {}
-        for i, resnet in enumerate(self.resnets):
-            x, new_cache[f"resnet_{i}"] = resnet(x, temb, zq, conv_cache=conv_cache.get(f"resnet_{i}"))
-        return x, new_cache
-
-
-class UpBlock3D(nn.Module):
-    def __init__(self, in_channels, out_channels, temb_channels=0, num_layers=1,
-                 eps=1e-6, act_fn="silu", groups=32, spatial_norm_dim=16,
-                 add_upsample=True, compress_time=False, pad_mode="first"):
-        super().__init__()
-        self.resnets = nn.ModuleList([
-            ResnetBlock3D(
-                in_channels=in_channels if i == 0 else out_channels,
-                out_channels=out_channels,
-                temb_channels=temb_channels, groups=groups, eps=eps,
-                act_fn=act_fn, spatial_norm_dim=spatial_norm_dim, pad_mode=pad_mode,
-            )
-            for i in range(num_layers)
-        ])
-        self.upsamplers = nn.ModuleList([Upsample3D(out_channels, out_channels, compress_time=compress_time)]) if add_upsample else None
-
-    def forward(self, x, temb=None, zq=None, conv_cache=None):
-        new_cache = {}
-        conv_cache = conv_cache or {}
-        for i, resnet in enumerate(self.resnets):
-            x, new_cache[f"resnet_{i}"] = resnet(x, temb, zq, conv_cache=conv_cache.get(f"resnet_{i}"))
-        if self.upsamplers is not None:
-            for us in self.upsamplers:
-                x = us(x)
-        return x, new_cache
-
-
-class Encoder3D(nn.Module):
-    def __init__(self, in_channels=3, out_channels=16,
-                 block_out_channels=(128, 256, 256, 512),
-                 layers_per_block=3, act_fn="silu",
-                 eps=1e-6, groups=32, pad_mode="first",
-                 temporal_compression_ratio=4):
-        super().__init__()
-        temporal_compress_level = int(np.log2(temporal_compression_ratio))
-
-        self.conv_in = CausalConv3d(in_channels, block_out_channels[0], kernel_size=3, pad_mode=pad_mode)
-
-        self.down_blocks = nn.ModuleList()
-        output_channel = block_out_channels[0]
-        for i in range(len(block_out_channels)):
-            input_channel = output_channel
-            output_channel = block_out_channels[i]
-            is_final = i == len(block_out_channels) - 1
-            compress_time = i < temporal_compress_level
-
-            self.down_blocks.append(DownBlock3D(
-                in_channels=input_channel, out_channels=output_channel,
-                temb_channels=0, num_layers=layers_per_block,
-                eps=eps, act_fn=act_fn, groups=groups,
-                add_downsample=not is_final, compress_time=compress_time,
-            ))
-
-        self.mid_block = MidBlock3D(
-            in_channels=block_out_channels[-1], temb_channels=0,
-            num_layers=2, eps=eps, act_fn=act_fn, groups=groups, pad_mode=pad_mode,
-        )
-
-        self.norm_out = ops.GroupNorm(groups, block_out_channels[-1], eps=1e-6)
-        self.conv_act = nn.SiLU()
-        self.conv_out = CausalConv3d(block_out_channels[-1], 2 * out_channels, kernel_size=3, pad_mode=pad_mode)
-
-    def forward(self, x, conv_cache=None):
-        new_cache = {}
-        conv_cache = conv_cache or {}
-
-        x, new_cache["conv_in"] = self.conv_in(x, conv_cache=conv_cache.get("conv_in"))
-
-        for i, block in enumerate(self.down_blocks):
-            key = f"down_block_{i}"
-            x, new_cache[key] = block(x, None, None, conv_cache.get(key))
-
-        x, new_cache["mid_block"] = self.mid_block(x, None, None, conv_cache=conv_cache.get("mid_block"))
-
-        x = self.norm_out(x)
-        x = self.conv_act(x)
-        x, new_cache["conv_out"] = self.conv_out(x, conv_cache=conv_cache.get("conv_out"))
-
-        return x, new_cache
-
-
-class Decoder3D(nn.Module):
-    def __init__(self, in_channels=16, out_channels=3,
-                 block_out_channels=(128, 256, 256, 512),
-                 layers_per_block=3, act_fn="silu",
-                 eps=1e-6, groups=32, pad_mode="first",
-                 temporal_compression_ratio=4):
-        super().__init__()
-        reversed_channels = list(reversed(block_out_channels))
-        temporal_compress_level = int(np.log2(temporal_compression_ratio))
-
-        self.conv_in = CausalConv3d(in_channels, reversed_channels[0], kernel_size=3, pad_mode=pad_mode)
-
-        self.mid_block = MidBlock3D(
-            in_channels=reversed_channels[0], temb_channels=0,
-            num_layers=2, eps=eps, act_fn=act_fn, groups=groups,
-            spatial_norm_dim=in_channels, pad_mode=pad_mode,
-        )
-
-        self.up_blocks = nn.ModuleList()
-        output_channel = reversed_channels[0]
-        for i in range(len(block_out_channels)):
-            prev_channel = output_channel
-            output_channel = reversed_channels[i]
-            is_final = i == len(block_out_channels) - 1
-            compress_time = i < temporal_compress_level
-
-            self.up_blocks.append(UpBlock3D(
-                in_channels=prev_channel, out_channels=output_channel,
-                temb_channels=0, num_layers=layers_per_block + 1,
-                eps=eps, act_fn=act_fn, groups=groups,
-                spatial_norm_dim=in_channels,
-                add_upsample=not is_final, compress_time=compress_time,
-            ))
-
-        self.norm_out = SpatialNorm3D(reversed_channels[-1], in_channels, groups=groups)
-        self.conv_act = nn.SiLU()
-        self.conv_out = CausalConv3d(reversed_channels[-1], out_channels, kernel_size=3, pad_mode=pad_mode)
-
-    def forward(self, sample, conv_cache=None):
-        new_cache = {}
-        conv_cache = conv_cache or {}
-
-        x, new_cache["conv_in"] = self.conv_in(sample, conv_cache=conv_cache.get("conv_in"))
-
-        x, new_cache["mid_block"] = self.mid_block(x, None, sample, conv_cache=conv_cache.get("mid_block"))
-
-        for i, block in enumerate(self.up_blocks):
-            key = f"up_block_{i}"
-            x, new_cache[key] = block(x, None, sample, conv_cache=conv_cache.get(key))
-
-        x, new_cache["norm_out"] = self.norm_out(x, sample, conv_cache=conv_cache.get("norm_out"))
-        x = self.conv_act(x)
-        x, new_cache["conv_out"] = self.conv_out(x, conv_cache=conv_cache.get("conv_out"))
-
-        return x, new_cache
-
-
-
-class AutoencoderKLCogVideoX(nn.Module):
-    """CogVideoX VAE. Spatial tiling/slicing handled by ComfyUI's VAE wrapper.
-
-    Uses rolling temporal decode: conv_in + mid_block + temporal up_blocks run
-    on the full (low-res) tensor, then the expensive spatial-only up_blocks +
-    norm_out + conv_out are processed in small temporal chunks with conv_cache
-    carrying causal state between chunks. This keeps peak VRAM proportional to
-    chunk_size rather than total frame count.
-    """
-
-    def __init__(self,
-                 in_channels=3, out_channels=3,
-                 block_out_channels=(128, 256, 256, 512),
-                 latent_channels=16, layers_per_block=3,
-                 act_fn="silu", eps=1e-6, groups=32,
-                 temporal_compression_ratio=4,
-                 ):
-        super().__init__()
-        self.latent_channels = latent_channels
-        self.temporal_compression_ratio = temporal_compression_ratio
-
-        self.encoder = Encoder3D(
-            in_channels=in_channels, out_channels=latent_channels,
-            block_out_channels=block_out_channels, layers_per_block=layers_per_block,
-            act_fn=act_fn, eps=eps, groups=groups,
-            temporal_compression_ratio=temporal_compression_ratio,
-        )
-        self.decoder = Decoder3D(
-            in_channels=latent_channels, out_channels=out_channels,
-            block_out_channels=block_out_channels, layers_per_block=layers_per_block,
-            act_fn=act_fn, eps=eps, groups=groups,
-            temporal_compression_ratio=temporal_compression_ratio,
-        )
-
-        self.num_latent_frames_batch_size = 2
-        self.num_sample_frames_batch_size = 8
-
-    def encode(self, x):
-        t = x.shape[2]
-        frame_batch = self.num_sample_frames_batch_size
-        remainder = t % frame_batch
-        conv_cache = None
-        enc = []
-
-        # Process remainder frames first so only the first chunk can have an
-        # odd temporal dimension — where Downsample3D's first-frame-special
-        # handling in temporal compression is actually correct.
-        if remainder > 0:
-            chunk, conv_cache = self.encoder(x[:, :, :remainder], conv_cache=conv_cache)
-            enc.append(chunk.to(x.device))
-
-        for start in range(remainder, t, frame_batch):
-            chunk, conv_cache = self.encoder(x[:, :, start:start + frame_batch], conv_cache=conv_cache)
-            enc.append(chunk.to(x.device))
-
-        enc = torch.cat(enc, dim=2)
-        mean, _ = enc.chunk(2, dim=1)
-        return mean
-
-    def decode(self, z):
-        return self._decode_rolling(z)
-
-    def _decode_batched(self, z):
-        """Original batched decode - processes 2 latent frames through full decoder."""
-        t = z.shape[2]
-        frame_batch = self.num_latent_frames_batch_size
-        num_batches = max(t // frame_batch, 1)
-        conv_cache = None
-        dec = []
-        for i in range(num_batches):
-            remaining = t % frame_batch
-            start = frame_batch * i + (0 if i == 0 else remaining)
-            end = frame_batch * (i + 1) + remaining
-            chunk, conv_cache = self.decoder(z[:, :, start:end], conv_cache=conv_cache)
-            dec.append(chunk.cpu())
-        return torch.cat(dec, dim=2).to(z.device)
-
-    def _decode_rolling(self, z):
-        """Rolling decode - processes low-res layers on full tensor, then rolls
-        through expensive high-res layers in temporal chunks."""
-        decoder = self.decoder
-        device = z.device
-
-        # Determine which up_blocks have temporal upsample vs spatial-only.
-        # Temporal up_blocks are cheap (low res), spatial-only are expensive.
-        temporal_compress_level = int(np.log2(self.temporal_compression_ratio))
-        split_at = temporal_compress_level  # first N up_blocks do temporal upsample
-
-        # Phase 1: conv_in + mid_block + temporal up_blocks on full tensor (low/medium res)
-        x, _ = decoder.conv_in(z)
-        x, _ = decoder.mid_block(x, None, z)
-
-        for i in range(split_at):
-            x, _ = decoder.up_blocks[i](x, None, z)
-
-        # Phase 2: remaining spatial-only up_blocks + norm_out + conv_out in temporal chunks
-        remaining_blocks = list(range(split_at, len(decoder.up_blocks)))
-        chunk_size = 4  # pixel frames per chunk through high-res layers
-        t_expanded = x.shape[2]
-
-        if t_expanded <= chunk_size or len(remaining_blocks) == 0:
-            # Small enough to process in one go
-            for i in remaining_blocks:
-                x, _ = decoder.up_blocks[i](x, None, z)
-            x, _ = decoder.norm_out(x, z)
-            x = decoder.conv_act(x)
-            x, _ = decoder.conv_out(x)
-            return x
-
-        # Expand z temporally once to match Phase 2's time dimension.
-        # z stays at latent spatial resolution so this is small (~16 MB vs ~1.3 GB
-        # for the old approach of pre-interpolating to every pixel resolution).
-        z_time_expanded = _interpolate_zq(z, (t_expanded, z.shape[3], z.shape[4]))
-
-        # Process in temporal chunks, interpolating spatially per-chunk to avoid
-        # allocating full [B, C, t_expanded, H, W] tensors at each resolution.
-        dec_out = []
-        conv_caches = {}
-
-        for chunk_start in range(0, t_expanded, chunk_size):
-            chunk_end = min(chunk_start + chunk_size, t_expanded)
-            x_chunk = x[:, :, chunk_start:chunk_end]
-            z_t_chunk = z_time_expanded[:, :, chunk_start:chunk_end]
-            z_spatial_cache = {}
-
-            for i in remaining_blocks:
-                block = decoder.up_blocks[i]
-                cache_key = f"up_block_{i}"
-                hw_key = (x_chunk.shape[3], x_chunk.shape[4])
-                if hw_key not in z_spatial_cache:
-                    if z_t_chunk.shape[3] == hw_key[0] and z_t_chunk.shape[4] == hw_key[1]:
-                        z_spatial_cache[hw_key] = z_t_chunk
-                    else:
-                        z_spatial_cache[hw_key] = F.interpolate(z_t_chunk, size=(z_t_chunk.shape[2], hw_key[0], hw_key[1]))
-                x_chunk, new_cache = block(x_chunk, None, z_spatial_cache[hw_key], conv_cache=conv_caches.get(cache_key))
-                conv_caches[cache_key] = new_cache
-
-            hw_key = (x_chunk.shape[3], x_chunk.shape[4])
-            if hw_key not in z_spatial_cache:
-                z_spatial_cache[hw_key] = F.interpolate(z_t_chunk, size=(z_t_chunk.shape[2], hw_key[0], hw_key[1]))
-            x_chunk, new_cache = decoder.norm_out(x_chunk, z_spatial_cache[hw_key], conv_cache=conv_caches.get("norm_out"))
-            conv_caches["norm_out"] = new_cache
-            x_chunk = decoder.conv_act(x_chunk)
-            x_chunk, new_cache = decoder.conv_out(x_chunk, conv_cache=conv_caches.get("conv_out"))
-            conv_caches["conv_out"] = new_cache
-
-            dec_out.append(x_chunk.cpu())
-            del z_spatial_cache
-
-        del x, z_time_expanded
-        return torch.cat(dec_out, dim=2).to(device)
--- a/comfy/lora.py
+++ b/comfy/lora.py
@ -342,12 +342,6 @@ def model_lora_keys_unet(model, key_map={}):
                key_map["base_model.model.{}".format(key_lora)] = k  # Official base model loras
                key_map["lycoris_{}".format(key_lora.replace(".", "_"))] = k  # LyCORIS/LoKR format

-    if isinstance(model, comfy.model_base.ErnieImage):
-        for k in sdk:
-            if k.startswith("diffusion_model.") and k.endswith(".weight"):
-                key_lora = k[len("diffusion_model."):-len(".weight")]
-                key_map["transformer.{}".format(key_lora)] = k
-
    return key_map


--- a/comfy/model_base.py
+++ b/comfy/model_base.py
@ -52,7 +52,6 @@ import comfy.ldm.qwen_image.model
 import comfy.ldm.kandinsky5.model
 import comfy.ldm.anima.model
 import comfy.ldm.ace.ace_step15
-import comfy.ldm.cogvideo.model
 import comfy.ldm.rt_detr.rtdetr_v4
 import comfy.ldm.ernie.model
 import comfy.ldm.sam3.detector
@ -82,7 +81,6 @@ class ModelType(Enum):
    IMG_TO_IMG = 9
    FLOW_COSMOS = 10
    IMG_TO_IMG_FLOW = 11
-    V_PREDICTION_DDPM = 12


 def model_sampling(model_config, model_type):
@ -117,8 +115,6 @@ def model_sampling(model_config, model_type):
        s = comfy.model_sampling.ModelSamplingCosmosRFlow
    elif model_type == ModelType.IMG_TO_IMG_FLOW:
        c = comfy.model_sampling.IMG_TO_IMG_FLOW
-    elif model_type == ModelType.V_PREDICTION_DDPM:
-        c = comfy.model_sampling.V_PREDICTION_DDPM

    class ModelSampling(s, c):
        pass
@ -1983,59 +1979,3 @@ class ErnieImage(BaseModel):
 class SAM3(BaseModel):
    def __init__(self, model_config, model_type=ModelType.FLOW, device=None):
        super().__init__(model_config, model_type, device=device, unet_model=comfy.ldm.sam3.detector.SAM3Model)
-
-class CogVideoX(BaseModel):
-    def __init__(self, model_config, model_type=ModelType.V_PREDICTION_DDPM, image_to_video=False, device=None):
-        super().__init__(model_config, model_type, device=device, unet_model=comfy.ldm.cogvideo.model.CogVideoXTransformer3DModel)
-        self.image_to_video = image_to_video
-
-    def concat_cond(self, **kwargs):
-        noise = kwargs.get("noise", None)
-        # Detect extra channels needed (e.g. 32 - 16 = 16 for ref latent)
-        extra_channels = self.diffusion_model.in_channels - noise.shape[1]
-        if extra_channels == 0:
-            return None
-
-        image = kwargs.get("concat_latent_image", None)
-        device = kwargs["device"]
-
-        if image is None:
-            shape = list(noise.shape)
-            shape[1] = extra_channels
-            return torch.zeros(shape, dtype=noise.dtype, layout=noise.layout, device=noise.device)
-
-        latent_dim = self.latent_format.latent_channels
-        image = utils.common_upscale(image.to(device), noise.shape[-1], noise.shape[-2], "bilinear", "center")
-
-        if noise.ndim == 5 and image.ndim == 5:
-            if image.shape[-3] < noise.shape[-3]:
-                image = torch.nn.functional.pad(image, (0, 0, 0, 0, 0, noise.shape[-3] - image.shape[-3]), "constant", 0)
-            elif image.shape[-3] > noise.shape[-3]:
-                image = image[:, :, :noise.shape[-3]]
-
-        for i in range(0, image.shape[1], latent_dim):
-            image[:, i:i + latent_dim] = self.process_latent_in(image[:, i:i + latent_dim])
-        image = utils.resize_to_batch_size(image, noise.shape[0])
-
-        if image.shape[1] > extra_channels:
-            image = image[:, :extra_channels]
-        elif image.shape[1] < extra_channels:
-            repeats = extra_channels // image.shape[1]
-            remainder = extra_channels % image.shape[1]
-            parts = [image] * repeats
-            if remainder > 0:
-                parts.append(image[:, :remainder])
-            image = torch.cat(parts, dim=1)
-
-        return image
-
-    def extra_conds(self, **kwargs):
-        out = super().extra_conds(**kwargs)
-        # OFS embedding (CogVideoX 1.5 I2V), default 2.0 as used by SparkVSR
-        if self.diffusion_model.ofs_proj_dim is not None:
-            ofs = kwargs.get("ofs", None)
-            if ofs is None:
-                noise = kwargs.get("noise", None)
-                ofs = torch.full((noise.shape[0],), 2.0, device=noise.device, dtype=noise.dtype)
-            out['ofs'] = comfy.conds.CONDRegular(ofs)
-        return out
--- a/comfy/model_detection.py
+++ b/comfy/model_detection.py
@ -490,54 +490,6 @@ def detect_unet_config(state_dict, key_prefix, metadata=None):

        return dit_config

-    if '{}blocks.0.norm1.linear.weight'.format(key_prefix) in state_dict_keys:  # CogVideoX
-        dit_config = {}
-        dit_config["image_model"] = "cogvideox"
-
-        # Extract config from weight shapes
-        norm1_weight = state_dict['{}blocks.0.norm1.linear.weight'.format(key_prefix)]
-        time_embed_dim = norm1_weight.shape[1]
-        dim = norm1_weight.shape[0] // 6
-
-        dit_config["num_attention_heads"] = dim // 64
-        dit_config["attention_head_dim"] = 64
-        dit_config["time_embed_dim"] = time_embed_dim
-        dit_config["num_layers"] = count_blocks(state_dict_keys, '{}blocks.'.format(key_prefix) + '{}.')
-
-        # Detect in_channels from patch_embed
-        patch_proj_key = '{}patch_embed.proj.weight'.format(key_prefix)
-        if patch_proj_key in state_dict_keys:
-            w = state_dict[patch_proj_key]
-            if w.ndim == 4:
-                # Conv2d: [out, in, kh, kw] — CogVideoX 1.0
-                dit_config["in_channels"] = w.shape[1]
-                dit_config["patch_size"] = w.shape[2]
-            elif w.ndim == 2:
-                # Linear: [out, in_channels * patch_size * patch_size * patch_size_t] — CogVideoX 1.5
-                dit_config["patch_size"] = 2
-                dit_config["patch_size_t"] = 2
-                dit_config["in_channels"] = w.shape[1] // (2 * 2 * 2)  # 256 // 8 = 32
-
-        text_proj_key = '{}patch_embed.text_proj.weight'.format(key_prefix)
-        if text_proj_key in state_dict_keys:
-            dit_config["text_embed_dim"] = state_dict[text_proj_key].shape[1]
-
-        # Detect OFS embedding
-        ofs_key = '{}ofs_embedding_linear_1.weight'.format(key_prefix)
-        if ofs_key in state_dict_keys:
-            dit_config["ofs_embed_dim"] = state_dict[ofs_key].shape[1]
-
-        # Detect positional embedding type
-        pos_key = '{}patch_embed.pos_embedding'.format(key_prefix)
-        if pos_key in state_dict_keys:
-            dit_config["use_learned_positional_embeddings"] = True
-            dit_config["use_rotary_positional_embeddings"] = False
-        else:
-            dit_config["use_learned_positional_embeddings"] = False
-            dit_config["use_rotary_positional_embeddings"] = True
-
-        return dit_config
-
    if '{}head.modulation'.format(key_prefix) in state_dict_keys:  # Wan 2.1
        dit_config = {}
        dit_config["image_model"] = "wan2.1"
@ -905,16 +857,9 @@ def detect_unet_config(state_dict, key_prefix, metadata=None):
    return unet_config

 def model_config_from_unet_config(unet_config, state_dict=None):
-    best_match = None
-    best_score = -1
    for model_config in comfy.supported_models.models:
-        score = model_config.match_score(unet_config, state_dict)
-        if score > best_score:
-            best_score = score
-            best_match = model_config
-
-    if best_match is not None:
-        return best_match(unet_config)
+        if model_config.matches(unet_config, state_dict):
+            return model_config(unet_config)

    logging.error("no match {}".format(unet_config))
    return None
--- a/comfy/model_management.py
+++ b/comfy/model_management.py
@ -663,7 +663,6 @@ def minimum_inference_memory():

 def free_memory(memory_required, device, keep_loaded=[], for_dynamic=False, pins_required=0, ram_required=0):
    cleanup_models_gc()
-    comfy.memory_management.extra_ram_release(max(pins_required, ram_required))
    unloaded_model = []
    can_unload = []
    unloaded_models = []
--- a/comfy/model_sampling.py
+++ b/comfy/model_sampling.py
@ -54,30 +54,6 @@ class V_PREDICTION(EPS):
        sigma = reshape_sigma(sigma, model_output.ndim)
        return model_input * self.sigma_data ** 2 / (sigma ** 2 + self.sigma_data ** 2) - model_output * sigma * self.sigma_data / (sigma ** 2 + self.sigma_data ** 2) ** 0.5

-class V_PREDICTION_DDPM:
-    """CogVideoX v-prediction: model receives raw x_t (unscaled), predicts velocity v.
-    x_0 = sqrt(alpha) * x_t - sqrt(1-alpha) * v
-        = x_t / sqrt(sigma^2 + 1) - v * sigma / sqrt(sigma^2 + 1)
-    """
-    def calculate_input(self, sigma, noise):
-        return noise
-
-    def calculate_denoised(self, sigma, model_output, model_input):
-        sigma = reshape_sigma(sigma, model_output.ndim)
-        return model_input / (sigma ** 2 + 1.0) ** 0.5 - model_output * sigma / (sigma ** 2 + 1.0) ** 0.5
-
-    def noise_scaling(self, sigma, noise, latent_image, max_denoise=False):
-        sigma = reshape_sigma(sigma, noise.ndim)
-        if max_denoise:
-            noise = noise * torch.sqrt(1.0 + sigma ** 2.0)
-        else:
-            noise = noise * sigma
-        noise += latent_image
-        return noise
-
-    def inverse_noise_scaling(self, sigma, latent):
-        return latent
-
 class EDM(V_PREDICTION):
    def calculate_denoised(self, sigma, model_output, model_input):
        sigma = reshape_sigma(sigma, model_output.ndim)
--- a/comfy/pinned_memory.py
+++ b/comfy/pinned_memory.py
@ -2,6 +2,7 @@ import comfy.model_management
 import comfy.memory_management
 import comfy_aimdo.host_buffer
 import comfy_aimdo.torch
+import psutil

 from comfy.cli_args import args

@ -11,6 +12,11 @@ def get_pin(module):
 def pin_memory(module):
    if module.pin_failed or args.disable_pinned_memory or get_pin(module) is not None:
        return
+    #FIXME: This is a RAM cache trigger event
+    ram_headroom = comfy.memory_management.RAM_CACHE_HEADROOM
+    #we split the difference and assume half the RAM cache headroom is for us
+    if ram_headroom > 0 and psutil.virtual_memory().available < (ram_headroom * 0.5):
+        comfy.memory_management.extra_ram_release(ram_headroom)

    size = comfy.memory_management.vram_aligned_size([ module.weight, module.bias ])

--- a/comfy/sd.py
+++ b/comfy/sd.py
@ -18,7 +18,6 @@ import comfy.ldm.wan.vae
 import comfy.ldm.wan.vae2_2
 import comfy.ldm.hunyuan3d.vae
 import comfy.ldm.ace.vae.music_dcae_pipeline
-import comfy.ldm.cogvideo.vae
 import comfy.ldm.hunyuan_video.vae
 import comfy.ldm.mmaudio.vae.autoencoder
 import comfy.pixel_space_convert
@ -479,10 +478,7 @@ class VAE:
                                                            encoder_config={'target': "comfy.ldm.modules.diffusionmodules.model.Encoder", 'params': encoder_config},
                                                            decoder_config={'target': "comfy.ldm.modules.temporal_ae.VideoDecoder", 'params': decoder_config})
            elif "taesd_decoder.1.weight" in sd:
-                if isinstance(metadata, dict) and "tae_latent_channels" in metadata:
-                    self.latent_channels = metadata["tae_latent_channels"]
-                else:
-                    self.latent_channels = sd["taesd_decoder.1.weight"].shape[1]
+                self.latent_channels = sd["taesd_decoder.1.weight"].shape[1]
                self.first_stage_model = comfy.taesd.taesd.TAESD(latent_channels=self.latent_channels)
            elif "vquantizer.codebook.weight" in sd: #VQGan: stage a of stable cascade
                self.first_stage_model = StageA()
@ -656,17 +652,6 @@ class VAE:

                self.memory_used_encode = lambda shape, dtype: (1400 * 9 * shape[-2] * shape[-1]) * model_management.dtype_size(dtype)
                self.memory_used_decode = lambda shape, dtype: (3600 * 4 * shape[-2] * shape[-1] * 16 * 16) * model_management.dtype_size(dtype)
-            elif "decoder.conv_in.conv.weight" in sd and "decoder.mid_block.resnets.0.norm1.norm_layer.weight" in sd:  # CogVideoX VAE
-                self.upscale_ratio = (lambda a: max(0, a * 4 - 3), 8, 8)
-                self.upscale_index_formula = (4, 8, 8)
-                self.downscale_ratio = (lambda a: max(0, math.floor((a + 3) / 4)), 8, 8)
-                self.downscale_index_formula = (4, 8, 8)
-                self.latent_dim = 3
-                self.latent_channels = sd["encoder.conv_out.conv.weight"].shape[0] // 2
-                self.first_stage_model = comfy.ldm.cogvideo.vae.AutoencoderKLCogVideoX(latent_channels=self.latent_channels)
-                self.memory_used_decode = lambda shape, dtype: (2800 * max(2, ((shape[2] - 1) * 4) + 1) * shape[3] * shape[4] * (8 * 8)) * model_management.dtype_size(dtype)
-                self.memory_used_encode = lambda shape, dtype: (1400 * max(1, shape[2]) * shape[3] * shape[4]) * model_management.dtype_size(dtype)
-                self.working_dtypes = [torch.bfloat16, torch.float16, torch.float32]
            elif "decoder.conv_in.conv.weight" in sd:
                ddconfig = {'double_z': True, 'z_channels': 4, 'resolution': 256, 'in_channels': 3, 'out_ch': 3, 'ch': 128, 'ch_mult': [1, 2, 4, 4], 'num_res_blocks': 2, 'attn_resolutions': [], 'dropout': 0.0}
                ddconfig["conv3d"] = True
--- a/comfy/supported_models.py
+++ b/comfy/supported_models.py
@ -27,7 +27,6 @@ import comfy.text_encoders.anima
 import comfy.text_encoders.ace15
 import comfy.text_encoders.longcat_image
 import comfy.text_encoders.ernie
-import comfy.text_encoders.cogvideo

 from . import supported_models_base
 from . import latent_formats
@ -1833,132 +1832,6 @@ class SAM31(SAM3):
    unet_config = {"image_model": "SAM31"}


-class CogVideoX_T2V(supported_models_base.BASE):
-    unet_config = {
-        "image_model": "cogvideox",
-    }
+models = [LotusD, Stable_Zero123, SD15_instructpix2pix, SD15, SD20, SD21UnclipL, SD21UnclipH, SDXL_instructpix2pix, SDXLRefiner, SDXL, SSD1B, KOALA_700M, KOALA_1B, Segmind_Vega, SD_X4Upscaler, Stable_Cascade_C, Stable_Cascade_B, SV3D_u, SV3D_p, SD3, StableAudio, AuraFlow, PixArtAlpha, PixArtSigma, HunyuanDiT, HunyuanDiT1, FluxInpaint, Flux, LongCatImage, FluxSchnell, GenmoMochi, LTXV, LTXAV, HunyuanVideo15_SR_Distilled, HunyuanVideo15, HunyuanImage21Refiner, HunyuanImage21, HunyuanVideoSkyreelsI2V, HunyuanVideoI2V, HunyuanVideo, CosmosT2V, CosmosI2V, CosmosT2IPredict2, CosmosI2VPredict2, ZImagePixelSpace, ZImage, Lumina2, WAN22_T2V, WAN21_T2V, WAN21_I2V, WAN21_FunControl2V, WAN21_Vace, WAN21_Camera, WAN22_Camera, WAN22_S2V, WAN21_HuMo, WAN22_Animate, WAN21_FlowRVS, WAN21_SCAIL, Hunyuan3Dv2mini, Hunyuan3Dv2, Hunyuan3Dv2_1, HiDream, Chroma, ChromaRadiance, ACEStep, ACEStep15, Omnigen2, QwenImage, Flux2, Kandinsky5Image, Kandinsky5, Anima, RT_DETR_v4, ErnieImage, SAM3, SAM31]

-    sampling_settings = {
-        "linear_start": 0.00085,
-        "linear_end": 0.012,
-        "beta_schedule": "linear",
-        "zsnr": True,
-    }
-
-    unet_extra_config = {}
-    latent_format = latent_formats.CogVideoX
-
-    supported_inference_dtypes = [torch.bfloat16, torch.float16, torch.float32]
-
-    vae_key_prefix = ["vae."]
-    text_encoder_key_prefix = ["text_encoders."]
-
-    def get_model(self, state_dict, prefix="", device=None):
-        # CogVideoX 1.5 (patch_size_t=2) has different training base dimensions for RoPE
-        if self.unet_config.get("patch_size_t") is not None:
-            self.unet_config.setdefault("sample_height", 96)
-            self.unet_config.setdefault("sample_width", 170)
-            self.unet_config.setdefault("sample_frames", 81)
-        out = model_base.CogVideoX(self, device=device)
-        return out
-
-    def clip_target(self, state_dict={}):
-        return supported_models_base.ClipTarget(comfy.text_encoders.cogvideo.CogVideoXT5Tokenizer, comfy.text_encoders.sd3_clip.T5XXLModel)
-
-class CogVideoX_I2V(CogVideoX_T2V):
-    unet_config = {
-        "image_model": "cogvideox",
-        "in_channels": 32,
-    }
-
-    def get_model(self, state_dict, prefix="", device=None):
-        if self.unet_config.get("patch_size_t") is not None:
-            self.unet_config.setdefault("sample_height", 96)
-            self.unet_config.setdefault("sample_width", 170)
-            self.unet_config.setdefault("sample_frames", 81)
-        out = model_base.CogVideoX(self, image_to_video=True, device=device)
-        return out
-
-
-models = [
-    LotusD,
-    Stable_Zero123,
-    SD15_instructpix2pix,
-    SD15,
-    SD20,
-    SD21UnclipL,
-    SD21UnclipH,
-    SDXL_instructpix2pix,
-    SDXLRefiner,
-    SDXL,
-    SSD1B,
-    KOALA_700M,
-    KOALA_1B,
-    Segmind_Vega,
-    SD_X4Upscaler,
-    Stable_Cascade_C,
-    Stable_Cascade_B,
-    SV3D_u,
-    SV3D_p,
-    SD3,
-    StableAudio,
-    AuraFlow,
-    PixArtAlpha,
-    PixArtSigma,
-    HunyuanDiT,
-    HunyuanDiT1,
-    FluxInpaint,
-    Flux,
-    LongCatImage,
-    FluxSchnell,
-    GenmoMochi,
-    LTXV,
-    LTXAV,
-    HunyuanVideo15_SR_Distilled,
-    HunyuanVideo15,
-    HunyuanImage21Refiner,
-    HunyuanImage21,
-    HunyuanVideoSkyreelsI2V,
-    HunyuanVideoI2V,
-    HunyuanVideo,
-    CosmosT2V,
-    CosmosI2V,
-    CosmosT2IPredict2,
-    CosmosI2VPredict2,
-    ZImagePixelSpace,
-    ZImage,
-    Lumina2,
-    WAN22_T2V,
-    WAN21_T2V,
-    WAN21_I2V,
-    WAN21_FunControl2V,
-    WAN21_Vace,
-    WAN21_Camera,
-    WAN22_Camera,
-    WAN22_S2V,
-    WAN21_HuMo,
-    WAN22_Animate,
-    WAN21_FlowRVS,
-    WAN21_SCAIL,
-    Hunyuan3Dv2mini,
-    Hunyuan3Dv2,
-    Hunyuan3Dv2_1,
-    HiDream,
-    Chroma,
-    ChromaRadiance,
-    ACEStep,
-    ACEStep15,
-    Omnigen2,
-    QwenImage,
-    Flux2,
-    Kandinsky5Image,
-    Kandinsky5,
-    Anima,
-    RT_DETR_v4,
-    ErnieImage,
-    SAM3,
-    SAM31,
-    CogVideoX_I2V,
-    CogVideoX_T2V,
-    SVD_img2vid,
-]
+models += [SVD_img2vid]
--- a/comfy/supported_models_base.py
+++ b/comfy/supported_models_base.py
@ -54,31 +54,15 @@ class BASE:
    optimizations = {"fp8": False}

    @classmethod
-    def match_score(s, unet_config, state_dict=None):
-        """Return a non-negative specificity score if this model matches the given
-        ``unet_config``/``state_dict``, otherwise ``-1``.
-
-        The score is the total number of ``unet_config`` keys (plus ``required_keys``
-        when ``state_dict`` is provided) that this model declares and that match the
-        input. Higher scores indicate a more specific match, allowing
-        :func:`model_config_from_unet_config` to prefer the most specific model when
-        several configs would otherwise match by subset.
-        """
-        score = 0
+    def matches(s, unet_config, state_dict=None):
        for k in s.unet_config:
            if k not in unet_config or s.unet_config[k] != unet_config[k]:
-                return -1
-            score += 1
+                return False
        if state_dict is not None:
            for k in s.required_keys:
                if k not in state_dict:
-                    return -1
-                score += 1
-        return score
-
-    @classmethod
-    def matches(s, unet_config, state_dict=None):
-        return s.match_score(unet_config, state_dict) >= 0
+                    return False
+        return True

    def model_type(self, state_dict, prefix=""):
        return model_base.ModelType.EPS
--- a/comfy/taesd/taehv.py
+++ b/comfy/taesd/taehv.py
@ -7,7 +7,6 @@ from tqdm.auto import tqdm
 from collections import namedtuple, deque

 import comfy.ops
-import comfy.model_management
 operations=comfy.ops.disable_weight_init

 DecoderResult = namedtuple("DecoderResult", ("frame", "memory"))
@ -48,14 +47,11 @@ class TGrow(nn.Module):
        x = self.conv(x)
        return x.reshape(-1, C, H, W)

-def apply_model_with_memblocks(model, x, parallel, show_progress_bar, output_device=None,
-                               patch_size=1, decode=False):
+def apply_model_with_memblocks(model, x, parallel, show_progress_bar):

    B, T, C, H, W = x.shape
    if parallel:
        x = x.reshape(B*T, C, H, W)
-        if not decode and patch_size > 1:
-            x = F.pixel_unshuffle(x, patch_size)
        # parallel over input timesteps, iterate over blocks
        for b in tqdm(model, disable=not show_progress_bar):
            if isinstance(b, MemBlock):
@ -66,27 +62,20 @@ def apply_model_with_memblocks(model, x, parallel, show_progress_bar, output_dev
                x = b(x, mem)
            else:
                x = b(x)
-        if decode and patch_size > 1:
-            x = F.pixel_shuffle(x, patch_size)
-        x = x.view(B, x.shape[0] // B, *x.shape[1:])
-        x = x.to(output_device)
+        BT, C, H, W = x.shape
+        T = BT // B
+        x = x.view(B, T, C, H, W)
    else:
        out = []
-        # Chunk along the time dim directly (chunks are [B,1,C,H,W] views, squeeze to [B,C,H,W] views).
-        # Avoids forcing a contiguous copy when x is non-contiguous (e.g. after movedim in encode/decode).
-        work_queue = deque([TWorkItem(xt.squeeze(1), 0) for xt in x.chunk(T, dim=1)])
+        work_queue = deque([TWorkItem(xt, 0) for t, xt in enumerate(x.reshape(B, T * C, H, W).chunk(T, dim=1))])
        progress_bar = tqdm(range(T), disable=not show_progress_bar)
        mem = [None] * len(model)
        while work_queue:
            xt, i = work_queue.popleft()
            if i == 0:
                progress_bar.update(1)
-                if not decode and patch_size > 1:
-                    xt = F.pixel_unshuffle(xt, patch_size)
            if i == len(model):
-                if decode and patch_size > 1:
-                    xt = F.pixel_shuffle(xt, patch_size)
-                out.append(xt.to(output_device))
+                out.append(xt)
                del xt
            else:
                b = model[i]
@ -176,20 +165,24 @@ class TAEHV(nn.Module):

    def encode(self, x, **kwargs):
        x = x.movedim(2, 1)  # [B, C, T, H, W] -> [B, T, C, H, W]
+        if self.patch_size > 1:
+            B, T, C, H, W = x.shape
+            x = x.reshape(B * T, C, H, W)
+            x = F.pixel_unshuffle(x, self.patch_size)
+            x = x.reshape(B, T, C * self.patch_size ** 2, H // self.patch_size, W // self.patch_size)
        if x.shape[1] % self.t_downscale != 0:
            # pad at end to multiple of t_downscale
            n_pad = self.t_downscale - x.shape[1] % self.t_downscale
            padding = x[:, -1:].repeat_interleave(n_pad, dim=1)
            x = torch.cat([x, padding], 1)
-        x = apply_model_with_memblocks(self.encoder, x, self.parallel, self.show_progress_bar,
-                                        patch_size=self.patch_size).movedim(2, 1)
+        x = apply_model_with_memblocks(self.encoder, x, self.parallel, self.show_progress_bar).movedim(2, 1)
        return self.process_out(x)

    def decode(self, x, **kwargs):
        x = x.unsqueeze(0) if x.ndim == 4 else x  # [T, C, H, W] -> [1, T, C, H, W]
        x = x.movedim(1, 2) if x.shape[1] != self.latent_channels else x  # [B, T, C, H, W] or [B, C, T, H, W]
        x = self.process_in(x).movedim(2, 1)  # [B, C, T, H, W] -> [B, T, C, H, W]
-        x = apply_model_with_memblocks(self.decoder, x, self.parallel, self.show_progress_bar,
-                                        output_device=comfy.model_management.intermediate_device(),
-                                        patch_size=self.patch_size, decode=True)
+        x = apply_model_with_memblocks(self.decoder, x, self.parallel, self.show_progress_bar)
+        if self.patch_size > 1:
+            x = F.pixel_shuffle(x, self.patch_size)
        return x[:, self.frames_to_trim:].movedim(2, 1)
--- a/comfy/taesd/taesd.py
+++ b/comfy/taesd/taesd.py
@ -17,79 +17,32 @@ class Clamp(nn.Module):
        return torch.tanh(x / 3) * 3

 class Block(nn.Module):
-    def __init__(self, n_in: int, n_out: int, use_midblock_gn: bool = False):
+    def __init__(self, n_in, n_out):
        super().__init__()
        self.conv = nn.Sequential(conv(n_in, n_out), nn.ReLU(), conv(n_out, n_out), nn.ReLU(), conv(n_out, n_out))
        self.skip = comfy.ops.disable_weight_init.Conv2d(n_in, n_out, 1, bias=False) if n_in != n_out else nn.Identity()
        self.fuse = nn.ReLU()
-        if not use_midblock_gn:
-            self.pool = None
-            return
-        n_gn = n_in * 4
-        self.pool = nn.Sequential(
-            comfy.ops.disable_weight_init.Conv2d(n_in, n_gn, 1, bias=False),
-            comfy.ops.disable_weight_init.GroupNorm(4, n_gn),
-            nn.ReLU(inplace=True),
-            comfy.ops.disable_weight_init.Conv2d(n_gn, n_in, 1, bias=False),
-        )
-
-    def forward(self, x: torch.Tensor) -> torch.Tensor:
-        if self.pool is not None:
-            x = x + self.pool(x)
+    def forward(self, x):
        return self.fuse(self.conv(x) + self.skip(x))

-class Encoder(nn.Sequential):
-    def __init__(self, latent_channels: int = 4, use_gn: bool = False):
-        super().__init__(
-            conv(3, 64), Block(64, 64),
-            conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
-            conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
-            conv(64, 64, stride=2, bias=False), Block(64, 64, use_gn), Block(64, 64, use_gn), Block(64, 64, use_gn),
-            conv(64, latent_channels),
-        )
+def Encoder(latent_channels=4):
+    return nn.Sequential(
+        conv(3, 64), Block(64, 64),
+        conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
+        conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
+        conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
+        conv(64, latent_channels),
+    )

-class Decoder(nn.Sequential):
-    def __init__(self, latent_channels: int = 4, use_gn: bool = False):
-        super().__init__(
-            Clamp(), conv(latent_channels, 64), nn.ReLU(),
-            Block(64, 64, use_gn), Block(64, 64, use_gn), Block(64, 64, use_gn), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
-            Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
-            Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
-            Block(64, 64), conv(64, 3),
-        )
-
-class DecoderFlux2(Decoder):
-    def __init__(self, latent_channels: int = 128, use_gn: bool = True):
-        if latent_channels != 128 or not use_gn:
-            raise ValueError("Unexpected parameters for Flux2 TAE module")
-        super().__init__(latent_channels=32, use_gn=True)
-
-    def forward(self, x: torch.Tensor) -> torch.Tensor:
-        B, C, H, W = x.shape
-        x = (
-            x
-            .reshape(B, 32, 2, 2, H, W)
-            .permute(0, 1, 4, 2, 5, 3)
-            .reshape(B, 32, H * 2, W * 2)
-        )
-        return super().forward(x)
-
-class EncoderFlux2(Encoder):
-    def __init__(self, latent_channels: int = 128, use_gn: bool = True):
-        if latent_channels != 128 or not use_gn:
-            raise ValueError("Unexpected parameters for Flux2 TAE module")
-        super().__init__(latent_channels=32, use_gn=True)
-
-    def forward(self, x: torch.Tensor) -> torch.Tensor:
-        result = super().forward(x)
-        B, C, H, W = result.shape
-        return (
-            result
-            .reshape(B, C, H // 2, 2, W // 2, 2)
-            .permute(0, 1, 3, 5, 2, 4)
-            .reshape(B, 128, H // 2, W // 2)
-        )

+def Decoder(latent_channels=4):
+    return nn.Sequential(
+        Clamp(), conv(latent_channels, 64), nn.ReLU(),
+        Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
+        Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
+        Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
+        Block(64, 64), conv(64, 3),
+    )

 class TAESD(nn.Module):
    latent_magnitude = 3
@ -98,15 +51,8 @@ class TAESD(nn.Module):
    def __init__(self, encoder_path=None, decoder_path=None, latent_channels=4):
        """Initialize pretrained TAESD on the given device from the given checkpoints."""
        super().__init__()
-        if latent_channels == 128:
-            encoder_class = EncoderFlux2
-            decoder_class = DecoderFlux2
-        else:
-            encoder_class = Encoder
-            decoder_class = Decoder
-        self.taesd_encoder = encoder_class(latent_channels=latent_channels)
-        self.taesd_decoder = decoder_class(latent_channels=latent_channels)
-
+        self.taesd_encoder = Encoder(latent_channels=latent_channels)
+        self.taesd_decoder = Decoder(latent_channels=latent_channels)
        self.vae_scale = torch.nn.Parameter(torch.tensor(1.0))
        self.vae_shift = torch.nn.Parameter(torch.tensor(0.0))
        if encoder_path is not None:
@ -115,19 +61,19 @@ class TAESD(nn.Module):
            self.taesd_decoder.load_state_dict(comfy.utils.load_torch_file(decoder_path, safe_load=True))

    @staticmethod
-    def scale_latents(x: torch.Tensor) -> torch.Tensor:
+    def scale_latents(x):
        """raw latents -> [0, 1]"""
        return x.div(2 * TAESD.latent_magnitude).add(TAESD.latent_shift).clamp(0, 1)

    @staticmethod
-    def unscale_latents(x: torch.Tensor) -> torch.Tensor:
+    def unscale_latents(x):
        """[0, 1] -> raw latents"""
        return x.sub(TAESD.latent_shift).mul(2 * TAESD.latent_magnitude)

-    def decode(self, x: torch.Tensor) -> torch.Tensor:
+    def decode(self, x):
        x_sample = self.taesd_decoder((x - self.vae_shift) * self.vae_scale)
        x_sample = x_sample.sub(0.5).mul(2)
        return x_sample

-    def encode(self, x: torch.Tensor) -> torch.Tensor:
+    def encode(self, x):
        return (self.taesd_encoder(x * 0.5 + 0.5) / self.vae_scale) + self.vae_shift
--- a/comfy/text_encoders/cogvideo.py
+++ b/comfy/text_encoders/cogvideo.py
@ -1,6 +0,0 @@
-import comfy.text_encoders.sd3_clip
-
-
-class CogVideoXT5Tokenizer(comfy.text_encoders.sd3_clip.T5XXLTokenizer):
-    def __init__(self, embedding_directory=None, tokenizer_data={}):
-        super().__init__(embedding_directory=embedding_directory, tokenizer_data=tokenizer_data, min_length=226)
--- a/comfy_api/latest/_input_impl/video_types.py
+++ b/comfy_api/latest/_input_impl/video_types.py
@ -251,7 +251,6 @@ class VideoFromFile(VideoInput):
            container.seek(start_pts, stream=video_stream)

        image_format = 'gbrpf32le'
-        process_image_format = lambda a: a
        audio = None

        streams = [video_stream]
@ -284,31 +283,14 @@ class VideoFromFile(VideoInput):
                            break

                        if not checked_alpha:
-                            alpha_channel = False
                            for comp in frame.format.components:
-                                if comp.is_alpha or frame.format.name == "pal8":
+                                if comp.is_alpha:
                                    alphas = []
-                                    alpha_channel = True
-                                    break
-                            if frame.format.name in ("yuvj420p", "yuvj422p", "yuvj444p", "rgb24", "rgba", "pal8"):
-                                process_image_format = lambda a: a.float() / 255.0
-                                if alpha_channel:
-                                    image_format = 'rgba'
-                                else:
-                                    image_format = 'rgb24'
-                            else:
-                                process_image_format = lambda a: a
-                                if alpha_channel:
                                    image_format = 'gbrapf32le'
-                                else:
-                                    image_format = 'gbrpf32le'
-
+                                    break
                            checked_alpha = True

                        img = frame.to_ndarray(format=image_format)  # shape: (H, W, 4)
-                        if frame.rotation != 0:
-                            k = int(round(frame.rotation // 90))
-                            img = np.rot90(img, k=k, axes=(0, 1)).copy()
                        if alphas is None:
                            frames.append(torch.from_numpy(img))
                        else:
@ -338,9 +320,9 @@ class VideoFromFile(VideoInput):
                    else:
                        audio_frames.append(frame.to_ndarray())

-        images = process_image_format(torch.stack(frames)) if len(frames) > 0 else torch.zeros(0, 0, 0, 3)
+        images = torch.stack(frames) if len(frames) > 0 else torch.zeros(0, 0, 0, 3)
        if alphas is not None:
-            alphas = process_image_format(torch.stack(alphas)) if len(alphas) > 0 else torch.zeros(0, 0, 0, 1)
+            alphas = torch.stack(alphas) if len(alphas) > 0 else torch.zeros(0, 0, 0, 1)

        # Get frame rate
        frame_rate = Fraction(video_stream.average_rate) if video_stream.average_rate else Fraction(1)
--- a/comfy_api_nodes/apis/bytedance.py
+++ b/comfy_api_nodes/apis/bytedance.py
@ -157,11 +157,6 @@ class SeedanceCreateAssetResponse(BaseModel):
    asset_id: str = Field(...)


-class SeedanceVirtualLibraryCreateAssetRequest(BaseModel):
-    url: str = Field(..., description="Publicly accessible URL of the image asset to upload.")
-    hash: str = Field(..., description="Dedup key. Re-submitting the same hash returns the existing asset id.")
-
-
 # Dollars per 1K tokens, keyed by (model_id, has_video_input).
 SEEDANCE2_PRICE_PER_1K_TOKENS = {
    ("dreamina-seedance-2-0-260128", False): 0.007,
--- a/comfy_api_nodes/nodes_bytedance.py
+++ b/comfy_api_nodes/nodes_bytedance.py
@ -1,4 +1,3 @@
-import hashlib
 import logging
 import math
 import re
@ -21,7 +20,6 @@ from comfy_api_nodes.apis.bytedance import (
    SeedanceCreateAssetResponse,
    SeedanceCreateVisualValidateSessionResponse,
    SeedanceGetVisualValidateSessionResponse,
-    SeedanceVirtualLibraryCreateAssetRequest,
    Seedream4Options,
    Seedream4TaskCreationRequest,
    TaskAudioContent,
@ -273,30 +271,6 @@ async def _wait_for_asset_active(cls: type[IO.ComfyNode], asset_id: str, group_i
    )


-async def _seedance_virtual_library_upload_image_asset(
-    cls: type[IO.ComfyNode],
-    image: torch.Tensor,
-    *,
-    wait_label: str = "Uploading image",
-) -> str:
-    """Upload an image into the caller's per-customer Seedance virtual library."""
-    public_url = await upload_image_to_comfyapi(cls, image, wait_label=wait_label)
-    normalized = image.detach().cpu().contiguous().to(torch.float32)
-    digest = hashlib.sha256()
-    digest.update(str(tuple(normalized.shape)).encode("utf-8"))
-    digest.update(b"\0")
-    digest.update(normalized.numpy().tobytes())
-    image_hash = digest.hexdigest()
-    create_resp = await sync_op(
-        cls,
-        ApiEndpoint(path="/proxy/seedance/virtual-library/assets", method="POST"),
-        response_model=SeedanceCreateAssetResponse,
-        data=SeedanceVirtualLibraryCreateAssetRequest(url=public_url, hash=image_hash),
-    )
-    await _wait_for_asset_active(cls, create_resp.asset_id, group_id="virtual-library")
-    return f"asset://{create_resp.asset_id}"
-
-
 def _seedance2_price_extractor(model_id: str, has_video_input: bool):
    """Returns a price_extractor closure for Seedance 2.0 poll_op."""
    rate = SEEDANCE2_PRICE_PER_1K_TOKENS.get((model_id, has_video_input))
@ -1533,9 +1507,7 @@ class ByteDance2FirstLastFrameNode(IO.ComfyNode):
        if first_frame_asset_id:
            first_frame_url = image_assets[first_frame_asset_id]
        else:
-            first_frame_url = await _seedance_virtual_library_upload_image_asset(
-                cls, first_frame, wait_label="Uploading first frame."
-            )
+            first_frame_url = await upload_image_to_comfyapi(cls, first_frame, wait_label="Uploading first frame.")

        content: list[TaskTextContent | TaskImageContent] = [
            TaskTextContent(text=model["prompt"]),
@ -1555,9 +1527,7 @@ class ByteDance2FirstLastFrameNode(IO.ComfyNode):
            content.append(
                TaskImageContent(
                    image_url=TaskImageContentUrl(
-                        url=await _seedance_virtual_library_upload_image_asset(
-                            cls, last_frame, wait_label="Uploading last frame."
-                        )
+                        url=await upload_image_to_comfyapi(cls, last_frame, wait_label="Uploading last frame.")
                    ),
                    role="last_frame",
                ),
@ -1835,9 +1805,9 @@ class ByteDance2ReferenceNode(IO.ComfyNode):
            content.append(
                TaskImageContent(
                    image_url=TaskImageContentUrl(
-                        url=await _seedance_virtual_library_upload_image_asset(
+                        url=await upload_image_to_comfyapi(
                            cls,
-                            reference_images[key],
+                            image=reference_images[key],
                            wait_label=f"Uploading image {i}",
                        ),
                    ),
--- a/comfy_api_nodes/nodes_openai.py
+++ b/comfy_api_nodes/nodes_openai.py
@ -415,9 +415,8 @@ class OpenAIGPTImage1(IO.ComfyNode):
                        "1152x2048",
                        "3840x2160",
                        "2160x3840",
-                        "Custom",
                    ],
-                    tooltip="Image size. Select 'Custom' to use the custom width and height (GPT Image 2 only).",
+                    tooltip="Image size",
                    optional=True,
                ),
                IO.Int.Input(
@ -446,24 +445,6 @@ class OpenAIGPTImage1(IO.ComfyNode):
                    default="gpt-image-2",
                    optional=True,
                ),
-                IO.Int.Input(
-                    "custom_width",
-                    default=1024,
-                    min=1024,
-                    max=3840,
-                    step=16,
-                    tooltip="Used only when `size` is 'Custom'. Must be a multiple of 16 (GPT Image 2 only).",
-                    optional=True,
-                ),
-                IO.Int.Input(
-                    "custom_height",
-                    default=1024,
-                    min=1024,
-                    max=3840,
-                    step=16,
-                    tooltip="Used only when `size` is 'Custom'. Must be a multiple of 16 (GPT Image 2 only).",
-                    optional=True,
-                ),
            ],
            outputs=[
                IO.Image.Output(),
@ -490,9 +471,9 @@ class OpenAIGPTImage1(IO.ComfyNode):
                      "high":   [0.133, 0.22]
                    },
                    "gpt-image-2": {
-                      "low":    [0.0048, 0.019],
-                      "medium": [0.041, 0.168],
-                      "high":   [0.165, 0.67]
+                      "low":    [0.0048, 0.012],
+                      "medium": [0.041, 0.112],
+                      "high":   [0.165, 0.43]
                    }
                  };
                  $range := $lookup($lookup($ranges, widgets.model), widgets.quality);
@ -522,8 +503,6 @@ class OpenAIGPTImage1(IO.ComfyNode):
        mask: Input.Image | None = None,
        n: int = 1,
        size: str = "1024x1024",
-        custom_width: int = 1024,
-        custom_height: int = 1024,
        model: str = "gpt-image-1",
    ) -> IO.NodeOutput:
        validate_string(prompt, strip_whitespace=False)
@ -531,25 +510,7 @@ class OpenAIGPTImage1(IO.ComfyNode):
        if mask is not None and image is None:
            raise ValueError("Cannot use a mask without an input image")

-        if size == "Custom":
-            if model != "gpt-image-2":
-                raise ValueError("Custom resolution is only supported by GPT Image 2 model")
-            if custom_width % 16 != 0 or custom_height % 16 != 0:
-                raise ValueError(f"Custom width and height must be multiples of 16, got {custom_width}x{custom_height}")
-            if max(custom_width, custom_height) > 3840:
-                raise ValueError(f"Custom resolution max edge must be <= 3840, got {custom_width}x{custom_height}")
-            ratio = max(custom_width, custom_height) / min(custom_width, custom_height)
-            if ratio > 3:
-                raise ValueError(
-                    f"Custom resolution aspect ratio must not exceed 3:1, got {custom_width}x{custom_height}"
-                )
-            total_pixels = custom_width * custom_height
-            if not 655_360 <= total_pixels <= 8_294_400:
-                raise ValueError(
-                    f"Custom resolution total pixels must be between 655,360 and 8,294,400, got {total_pixels}"
-                )
-            size = f"{custom_width}x{custom_height}"
-        elif model in ("gpt-image-1", "gpt-image-1.5"):
+        if model in ("gpt-image-1", "gpt-image-1.5"):
            if size not in ("auto", "1024x1024", "1024x1536", "1536x1024"):
                raise ValueError(f"Resolution {size} is only supported by GPT Image 2 model")

--- a/comfy_execution/caching.py
+++ b/comfy_execution/caching.py
@ -5,7 +5,6 @@ import psutil
 import time
 import torch
 from typing import Sequence, Mapping, Dict
-from comfy.model_patcher import ModelPatcher
 from comfy_execution.graph import DynamicPrompt
 from abc import ABC, abstractmethod

@ -524,15 +523,13 @@ class RAMPressureCache(LRUCache):
        self.timestamps[self.cache_key_set.get_data_key(node_id)] = time.time()
        super().set_local(node_id, value)

-    def ram_release(self, target, free_active=False):
+    def ram_release(self, target):
        if psutil.virtual_memory().available >= target:
            return

        clean_list = []

        for key, cache_entry in self.cache.items():
-            if not free_active and self.used_generation[key] == self.generation:
-                continue
            oom_score =  RAM_CACHE_OLD_WORKFLOW_OOM_MULTIPLIER ** (self.generation - self.used_generation[key])

            ram_usage = RAM_CACHE_DEFAULT_RAM_USAGE
@ -545,9 +542,6 @@ class RAMPressureCache(LRUCache):
                        scan_list_for_ram_usage(output)
                    elif isinstance(output, torch.Tensor) and output.device.type == 'cpu':
                        ram_usage += output.numel() * output.element_size()
-                    elif isinstance(output, ModelPatcher) and self.used_generation[key] != self.generation:
-                        #old ModelPatchers are the first to go
-                        ram_usage = 1e30
            scan_list_for_ram_usage(cache_entry.outputs)

            oom_score *= ram_usage
--- a/execution.py
+++ b/execution.py
@ -779,7 +779,7 @@ class PromptExecutor:

                    if self.cache_type == CacheType.RAM_PRESSURE:
                        comfy.model_management.free_memory(0, None, pins_required=ram_headroom, ram_required=ram_headroom)
-                        ram_release_callback(ram_headroom, free_active=True)
+                        comfy.memory_management.extra_ram_release(ram_headroom)
                else:
                    # Only execute when the while-loop ends without break
                    # Send cached UI for intermediate output nodes that weren't executed
--- a/nodes.py
+++ b/nodes.py
@ -32,7 +32,7 @@ import comfy.controlnet
 from comfy.comfy_types import IO, ComfyNodeABC, InputTypeDict, FileLocator
 from comfy_api.internal import register_versions, ComfyAPIWithVersion
 from comfy_api.version_list import supported_versions
-from comfy_api.latest import io, ComfyExtension, InputImpl
+from comfy_api.latest import io, ComfyExtension

 import comfy.clip_vision

@ -728,26 +728,50 @@ class LoraLoaderModelOnly(LoraLoader):

 class VAELoader:
    video_taes = ["taehv", "lighttaew2_2", "lighttaew2_1", "lighttaehy1_5", "taeltx_2"]
-    image_taes = ["taesd", "taesdxl", "taesd3", "taef1", "taef2"]
-
+    image_taes = ["taesd", "taesdxl", "taesd3", "taef1"]
    @staticmethod
    def vae_list(s):
        vaes = folder_paths.get_filename_list("vae")
        approx_vaes = folder_paths.get_filename_list("vae_approx")
-        have_img_encoder, have_img_decoder = set(), set()
+        sdxl_taesd_enc = False
+        sdxl_taesd_dec = False
+        sd1_taesd_enc = False
+        sd1_taesd_dec = False
+        sd3_taesd_enc = False
+        sd3_taesd_dec = False
+        f1_taesd_enc = False
+        f1_taesd_dec = False
+
        for v in approx_vaes:
-            parts = v.split("_", 1)
-            if len(parts) != 2 or parts[0] not in s.image_taes:
+            if v.startswith("taesd_decoder."):
+                sd1_taesd_dec = True
+            elif v.startswith("taesd_encoder."):
+                sd1_taesd_enc = True
+            elif v.startswith("taesdxl_decoder."):
+                sdxl_taesd_dec = True
+            elif v.startswith("taesdxl_encoder."):
+                sdxl_taesd_enc = True
+            elif v.startswith("taesd3_decoder."):
+                sd3_taesd_dec = True
+            elif v.startswith("taesd3_encoder."):
+                sd3_taesd_enc = True
+            elif v.startswith("taef1_encoder."):
+                f1_taesd_dec = True
+            elif v.startswith("taef1_decoder."):
+                f1_taesd_enc = True
+            else:
                for tae in s.video_taes:
                    if v.startswith(tae):
                        vaes.append(v)
-                        break
-                continue
-            if parts[1].startswith("encoder."):
-                have_img_encoder.add(parts[0])
-            elif parts[1].startswith("decoder."):
-                have_img_decoder.add(parts[0])
-        vaes += [k for k in have_img_decoder if k in have_img_encoder]
+
+        if sd1_taesd_dec and sd1_taesd_enc:
+            vaes.append("taesd")
+        if sdxl_taesd_dec and sdxl_taesd_enc:
+            vaes.append("taesdxl")
+        if sd3_taesd_dec and sd3_taesd_enc:
+            vaes.append("taesd3")
+        if f1_taesd_dec and f1_taesd_enc:
+            vaes.append("taef1")
        vaes.append("pixel_space")
        return vaes

@ -803,11 +827,6 @@ class VAELoader:
            else:
                vae_path = folder_paths.get_full_path_or_raise("vae", vae_name)
            sd, metadata = comfy.utils.load_torch_file(vae_path, return_metadata=True)
-        if vae_name == "taef2":
-            if metadata is None:
-                metadata = {"tae_latent_channels": 128}
-            else:
-                metadata["tae_latent_channels"] = 128
        vae = comfy.sd.VAE(sd=sd, metadata=metadata)
        vae.throw_exception_if_invalid()
        return (vae,)
@ -1697,10 +1716,6 @@ class LoadImage:
    def load_image(self, image):
        image_path = folder_paths.get_annotated_filepath(image)

-        components = InputImpl.VideoFromFile(image_path).get_components()
-        if components.images.shape[0] > 0:
-            return (components.images, 1.0 - components.alpha[..., -1] if components.alpha is not None else torch.zeros((components.images.shape[0], 64, 64), dtype=torch.float32, device="cpu"))
-
        img = node_helpers.pillow(Image.open, image_path)

        output_images = []
@ -2213,6 +2228,12 @@ async def load_custom_node(module_path: str, ignore=set(), module_parent="custom

        LOADED_MODULE_DIRS[module_name] = os.path.abspath(module_dir)

+        # Only load node_replacements.json from directory-based custom nodes (proper packs).
+        # Single-file .py nodes share a parent dir, so checking there would be incorrect.
+        if os.path.isdir(module_path):
+            from server import PromptServer
+            PromptServer.instance.node_replace_manager.load_from_json(module_dir, module_name)
+
        try:
            from comfy_config import config_parser

@ -2444,7 +2465,7 @@ async def init_builtin_extra_nodes():
        "nodes_curve.py",
        "nodes_rtdetr.py",
        "nodes_frame_interpolation.py",
-        "nodes_sam3.py",
+        "nodes_sam3.py"
    ]

    import_failed = []
--- a/requirements.txt
+++ b/requirements.txt
@ -1,5 +1,5 @@
 comfyui-frontend-package==1.42.15
-comfyui-workflow-templates==0.9.65
+comfyui-workflow-templates==0.9.63
 comfyui-embedded-docs==0.4.4
 torch
 torchsde
@ -19,11 +19,11 @@ scipy
 tqdm
 psutil
 alembic
-SQLAlchemy>=2.0.0
+SQLAlchemy>=2.0
 filelock
 av>=14.2.0
 comfy-kitchen>=0.2.8
-comfy-aimdo==0.3.0
+comfy-aimdo==0.2.14
 requests
 simpleeval>=1.0.0
 blake3
--- a/tests-unit/comfy_test/model_detection_test.py
+++ b/tests-unit/comfy_test/model_detection_test.py
@ -60,38 +60,18 @@ def _make_flux_schnell_comfyui_sd():


 class TestModelDetection:
-    """Verify that model detection selects the most specific model regardless of
-    the ordering of entries in ``comfy.supported_models.models``."""
+    """Verify that first-match model detection selects the correct model
+    based on list ordering and unet_config specificity."""

-    def test_longcat_detection_is_order_independent(self):
-        """Detection must pick LongCatImage over FluxSchnell regardless of
-        their relative order in the models list, because LongCatImage has a
-        strictly more specific ``unet_config``."""
-        original_models = comfy.supported_models.models
-        sd = _make_longcat_comfyui_sd()
-        unet_config = detect_unet_config(sd, "")
-
-        try:
-            for ordering in ("longcat_first", "schnell_first"):
-                models = list(original_models)
-                longcat = next(m for m in models if m.__name__ == "LongCatImage")
-                schnell = next(m for m in models if m.__name__ == "FluxSchnell")
-                models.remove(longcat)
-                models.remove(schnell)
-                if ordering == "longcat_first":
-                    models.extend([longcat, schnell])
-                else:
-                    models.extend([schnell, longcat])
-                comfy.supported_models.models = models
-
-                model_config = model_config_from_unet_config(unet_config, sd)
-                assert model_config is not None
-                assert type(model_config).__name__ == "LongCatImage", (
-                    f"Expected LongCatImage with ordering={ordering}, "
-                    f"got {type(model_config).__name__}"
-                )
-        finally:
-            comfy.supported_models.models = original_models
+    def test_longcat_before_schnell_in_models_list(self):
+        """LongCatImage must appear before FluxSchnell in the models list."""
+        models = comfy.supported_models.models
+        longcat_idx = next(i for i, m in enumerate(models) if m.__name__ == "LongCatImage")
+        schnell_idx = next(i for i, m in enumerate(models) if m.__name__ == "FluxSchnell")
+        assert longcat_idx < schnell_idx, (
+            f"LongCatImage (index {longcat_idx}) must come before "
+            f"FluxSchnell (index {schnell_idx}) in the models list"
+        )

    def test_longcat_comfyui_detected_as_longcat(self):
        sd = _make_longcat_comfyui_sd()
--- a/tests/test_node_replacements_json.py
+++ b/tests/test_node_replacements_json.py
@ -0,0 +1,217 @@
+"""Tests for NodeReplaceManager.load_from_json — auto-registration of
+node_replacements.json from custom node directories."""
+import json
+import os
+import tempfile
+import unittest
+
+from app.node_replace_manager import NodeReplaceManager
+
+
+class SimpleNodeReplace:
+    """Lightweight stand-in for comfy_api.latest._io.NodeReplace (avoids torch import)."""
+    def __init__(self, new_node_id, old_node_id, old_widget_ids=None,
+                 input_mapping=None, output_mapping=None):
+        self.new_node_id = new_node_id
+        self.old_node_id = old_node_id
+        self.old_widget_ids = old_widget_ids
+        self.input_mapping = input_mapping
+        self.output_mapping = output_mapping
+
+    def as_dict(self):
+        return {
+            "new_node_id": self.new_node_id,
+            "old_node_id": self.old_node_id,
+            "old_widget_ids": self.old_widget_ids,
+            "input_mapping": list(self.input_mapping) if self.input_mapping else None,
+            "output_mapping": list(self.output_mapping) if self.output_mapping else None,
+        }
+
+
+class TestLoadFromJson(unittest.TestCase):
+    """Test auto-registration of node_replacements.json from custom node directories."""
+
+    def setUp(self):
+        self.tmpdir = tempfile.mkdtemp()
+        self.manager = NodeReplaceManager()
+
+    def _write_json(self, data):
+        path = os.path.join(self.tmpdir, "node_replacements.json")
+        with open(path, "w") as f:
+            json.dump(data, f)
+
+    def _load(self):
+        self.manager.load_from_json(self.tmpdir, "test-node-pack", _node_replace_class=SimpleNodeReplace)
+
+    def test_no_file_does_nothing(self):
+        """No node_replacements.json — should silently do nothing."""
+        self._load()
+        self.assertEqual(self.manager.as_dict(), {})
+
+    def test_empty_object(self):
+        """Empty {} — should do nothing."""
+        self._write_json({})
+        self._load()
+        self.assertEqual(self.manager.as_dict(), {})
+
+    def test_single_replacement(self):
+        """Single replacement entry registers correctly."""
+        self._write_json({
+            "OldNode": [{
+                "new_node_id": "NewNode",
+                "old_node_id": "OldNode",
+                "input_mapping": [{"new_id": "model", "old_id": "ckpt_name"}],
+                "output_mapping": [{"new_idx": 0, "old_idx": 0}],
+            }]
+        })
+        self._load()
+        result = self.manager.as_dict()
+        self.assertIn("OldNode", result)
+        self.assertEqual(len(result["OldNode"]), 1)
+        entry = result["OldNode"][0]
+        self.assertEqual(entry["new_node_id"], "NewNode")
+        self.assertEqual(entry["old_node_id"], "OldNode")
+        self.assertEqual(entry["input_mapping"], [{"new_id": "model", "old_id": "ckpt_name"}])
+        self.assertEqual(entry["output_mapping"], [{"new_idx": 0, "old_idx": 0}])
+
+    def test_multiple_replacements(self):
+        """Multiple old_node_ids each with entries."""
+        self._write_json({
+            "NodeA": [{"new_node_id": "NodeB", "old_node_id": "NodeA"}],
+            "NodeC": [{"new_node_id": "NodeD", "old_node_id": "NodeC"}],
+        })
+        self._load()
+        result = self.manager.as_dict()
+        self.assertEqual(len(result), 2)
+        self.assertIn("NodeA", result)
+        self.assertIn("NodeC", result)
+
+    def test_multiple_alternatives_for_same_node(self):
+        """Multiple replacement options for the same old node."""
+        self._write_json({
+            "OldNode": [
+                {"new_node_id": "AltA", "old_node_id": "OldNode"},
+                {"new_node_id": "AltB", "old_node_id": "OldNode"},
+            ]
+        })
+        self._load()
+        result = self.manager.as_dict()
+        self.assertEqual(len(result["OldNode"]), 2)
+
+    def test_null_mappings(self):
+        """Null input/output mappings (trivial replacement)."""
+        self._write_json({
+            "OldNode": [{
+                "new_node_id": "NewNode",
+                "old_node_id": "OldNode",
+                "input_mapping": None,
+                "output_mapping": None,
+            }]
+        })
+        self._load()
+        entry = self.manager.as_dict()["OldNode"][0]
+        self.assertIsNone(entry["input_mapping"])
+        self.assertIsNone(entry["output_mapping"])
+
+    def test_old_node_id_defaults_to_key(self):
+        """If old_node_id is missing from entry, uses the dict key."""
+        self._write_json({
+            "OldNode": [{"new_node_id": "NewNode"}]
+        })
+        self._load()
+        entry = self.manager.as_dict()["OldNode"][0]
+        self.assertEqual(entry["old_node_id"], "OldNode")
+
+    def test_invalid_json_skips(self):
+        """Invalid JSON file — should warn and skip, not crash."""
+        path = os.path.join(self.tmpdir, "node_replacements.json")
+        with open(path, "w") as f:
+            f.write("{invalid json")
+        self._load()
+        self.assertEqual(self.manager.as_dict(), {})
+
+    def test_non_object_json_skips(self):
+        """JSON array instead of object — should warn and skip."""
+        self._write_json([1, 2, 3])
+        self._load()
+        self.assertEqual(self.manager.as_dict(), {})
+
+    def test_non_list_value_skips(self):
+        """Value is not a list — should warn and skip that key."""
+        self._write_json({
+            "OldNode": "not a list",
+            "GoodNode": [{"new_node_id": "NewNode", "old_node_id": "GoodNode"}],
+        })
+        self._load()
+        result = self.manager.as_dict()
+        self.assertNotIn("OldNode", result)
+        self.assertIn("GoodNode", result)
+
+    def test_with_old_widget_ids(self):
+        """old_widget_ids are passed through."""
+        self._write_json({
+            "OldNode": [{
+                "new_node_id": "NewNode",
+                "old_node_id": "OldNode",
+                "old_widget_ids": ["width", "height"],
+            }]
+        })
+        self._load()
+        entry = self.manager.as_dict()["OldNode"][0]
+        self.assertEqual(entry["old_widget_ids"], ["width", "height"])
+
+    def test_set_value_in_input_mapping(self):
+        """input_mapping with set_value entries."""
+        self._write_json({
+            "OldNode": [{
+                "new_node_id": "NewNode",
+                "old_node_id": "OldNode",
+                "input_mapping": [
+                    {"new_id": "method", "set_value": "lanczos"},
+                    {"new_id": "size", "old_id": "dimension"},
+                ],
+            }]
+        })
+        self._load()
+        entry = self.manager.as_dict()["OldNode"][0]
+        self.assertEqual(len(entry["input_mapping"]), 2)
+
+    def test_missing_new_node_id_skipped(self):
+        """Entry without new_node_id is skipped."""
+        self._write_json({
+            "OldNode": [
+                {"old_node_id": "OldNode"},
+                {"new_node_id": "", "old_node_id": "OldNode"},
+                {"new_node_id": "ValidNew", "old_node_id": "OldNode"},
+            ]
+        })
+        self._load()
+        result = self.manager.as_dict()
+        self.assertEqual(len(result["OldNode"]), 1)
+        self.assertEqual(result["OldNode"][0]["new_node_id"], "ValidNew")
+
+    def test_non_dict_entry_skipped(self):
+        """Non-dict entries in the list are silently skipped."""
+        self._write_json({
+            "OldNode": [
+                "not a dict",
+                {"new_node_id": "NewNode", "old_node_id": "OldNode"},
+            ]
+        })
+        self._load()
+        result = self.manager.as_dict()
+        self.assertEqual(len(result["OldNode"]), 1)
+
+    def test_has_replacement_after_load(self):
+        """Manager reports has_replacement correctly after JSON load."""
+        self._write_json({
+            "OldNode": [{"new_node_id": "NewNode", "old_node_id": "OldNode"}],
+        })
+        self.assertFalse(self.manager.has_replacement("OldNode"))
+        self._load()
+        self.assertTrue(self.manager.has_replacement("OldNode"))
+        self.assertFalse(self.manager.has_replacement("UnknownNode"))
+
+
+if __name__ == "__main__":
+    unittest.main()
Author	SHA1	Message	Date
Jedrzej Kosinski	5225f109a6	Merge branch 'master' into deepme987/auto-register-node-replacements-json	2026-04-28 02:53:37 -07:00
Deep Mehta	e35fe5bc09	Merge branch 'master' into deepme987/auto-register-node-replacements-json	2026-04-21 05:00:15 +05:30
Deep Mehta	77054cd49e	Merge branch 'master' into deepme987/auto-register-node-replacements-json	2026-04-14 19:34:21 -07:00
Deep Mehta	1cd2730b25	Merge branch 'master' into deepme987/auto-register-node-replacements-json	2026-04-06 13:13:42 -07:00
Deep Mehta	d4351f77f8	Merge branch 'master' into deepme987/auto-register-node-replacements-json	2026-03-25 22:50:44 -07:00
Deep Mehta	9837dd368a	refactor: move load_from_json into NodeReplaceManager Address review feedback from Kosinkadink: 1. Move JSON loading logic from nodes.py into NodeReplaceManager as load_from_json() method for better encapsulation and testability 2. Tests now exercise the real NodeReplaceManager (no duplicated logic) 3. Defer `import nodes` in apply_replacements to avoid torch at import 4. nodes.py call site simplified to one line: PromptServer.instance.node_replace_manager.load_from_json(...)	2026-03-25 22:12:23 -07:00
Deep Mehta	62ec9a3238	fix: skip single-file nodes and validate new_node_id Two fixes from code review: 1. Only load node_replacements.json from directory-based custom nodes. Single-file .py nodes share a parent dir (custom_nodes/), so checking there would incorrectly pick up a stray file. 2. Skip entries with missing or empty new_node_id instead of registering a replacement pointing to nothing.	2026-03-23 14:47:11 -07:00
Deep Mehta	b20cb7892e	Merge branch 'master' into deepme987/auto-register-node-replacements-json	2026-03-18 17:14:08 -07:00
Deep Mehta	b9b24d425b	Merge branch 'master' into deepme987/auto-register-node-replacements-json	2026-03-17 20:58:06 -07:00
Deep Mehta	d731cb6ae1	feat: auto-register node replacements from custom node JSON files Custom node authors can now ship a `node_replacements.json` in their repo root to define replacements declaratively. During node loading, ComfyUI reads these files and registers entries via the existing NodeReplaceManager — no Python registration code needed. This enables two use cases: 1. Authors deprecate/rename nodes with a migration path for old workflows 2. Authors offer their nodes as drop-in replacements for other packs	2026-03-17 20:57:32 -07:00