Merge 2db3ca609d into f5030e26fd

Add progress bar to ace step. (#12242 )
Simplified storing signatures/ancestors
2026-02-07 03:52:32 +08:00 · 2026-02-03 04:21:38 -05:00 · 2026-02-03 04:09:30 -05:00 · 2026-02-03 03:01:52 -05:00 · 2026-02-03 02:20:59 -05:00 · 2026-02-03 02:11:48 -05:00
24 changed files with 2267 additions and 144 deletions
--- a/comfy/latent_formats.py
+++ b/comfy/latent_formats.py
@ -755,6 +755,10 @@ class ACEAudio(LatentFormat):
    latent_channels = 8
    latent_dimensions = 2

+class ACEAudio15(LatentFormat):
+    latent_channels = 64
+    latent_dimensions = 1
+
 class ChromaRadiance(LatentFormat):
    latent_channels = 3
    spacial_downscale_ratio = 1
--- a/comfy/ldm/ace/ace_step15.py
+++ b/comfy/ldm/ace/ace_step15.py
--- a/comfy/model_base.py
+++ b/comfy/model_base.py
@ -50,6 +50,7 @@ import comfy.ldm.omnigen.omnigen2
 import comfy.ldm.qwen_image.model
 import comfy.ldm.kandinsky5.model
 import comfy.ldm.anima.model
+import comfy.ldm.ace.ace_step15

 import comfy.model_management
 import comfy.patcher_extension
@ -1540,6 +1541,47 @@ class ACEStep(BaseModel):
        out['lyrics_strength'] = comfy.conds.CONDConstant(kwargs.get("lyrics_strength", 1.0))
        return out

+class ACEStep15(BaseModel):
+    def __init__(self, model_config, model_type=ModelType.FLOW, device=None):
+        super().__init__(model_config, model_type, device=device, unet_model=comfy.ldm.ace.ace_step15.AceStepConditionGenerationModel)
+
+    def extra_conds(self, **kwargs):
+        out = super().extra_conds(**kwargs)
+        device = kwargs["device"]
+
+        cross_attn = kwargs.get("cross_attn", None)
+        if cross_attn is not None:
+            out['c_crossattn'] = comfy.conds.CONDRegular(cross_attn)
+
+        conditioning_lyrics = kwargs.get("conditioning_lyrics", None)
+        if cross_attn is not None:
+            out['lyric_embed'] = comfy.conds.CONDRegular(conditioning_lyrics)
+
+        refer_audio = kwargs.get("reference_audio_timbre_latents", None)
+        if refer_audio is None or len(refer_audio) == 0:
+            refer_audio = torch.tensor([[[-1.3672e-01, -1.5820e-01,  5.8594e-01, -5.7422e-01,  3.0273e-02,
+                                        2.7930e-01, -2.5940e-03, -2.0703e-01, -1.6113e-01, -1.4746e-01,
+                                        -2.7710e-02, -1.8066e-01, -2.9688e-01,  1.6016e+00, -2.6719e+00,
+                                        7.7734e-01, -1.3516e+00, -1.9434e-01, -7.1289e-02, -5.0938e+00,
+                                        2.4316e-01,  4.7266e-01,  4.6387e-02, -6.6406e-01, -2.1973e-01,
+                                        -6.7578e-01, -1.5723e-01,  9.5312e-01, -2.0020e-01, -1.7109e+00,
+                                        5.8984e-01, -5.7422e-01,  5.1562e-01,  2.8320e-01,  1.4551e-01,
+                                        -1.8750e-01, -5.9814e-02,  3.6719e-01, -1.0059e-01, -1.5723e-01,
+                                        2.0605e-01, -4.3359e-01, -8.2812e-01,  4.5654e-02, -6.6016e-01,
+                                        1.4844e-01,  9.4727e-02,  3.8477e-01, -1.2578e+00, -3.3203e-01,
+                                        -8.5547e-01,  4.3359e-01,  4.2383e-01, -8.9453e-01, -5.0391e-01,
+                                        -5.6152e-02, -2.9219e+00, -2.4658e-02,  5.0391e-01,  9.8438e-01,
+                                        7.2754e-02, -2.1582e-01,  6.3672e-01,  1.0000e+00]]], device=device).movedim(-1, 1).repeat(1, 1, 750)
+        else:
+            refer_audio = refer_audio[-1]
+        out['refer_audio'] = comfy.conds.CONDRegular(refer_audio)
+
+        audio_codes = kwargs.get("audio_codes", None)
+        if audio_codes is not None:
+            out['audio_codes'] = comfy.conds.CONDRegular(torch.tensor(audio_codes, device=device))
+
+        return out
+
 class Omnigen2(BaseModel):
    def __init__(self, model_config, model_type=ModelType.FLOW, device=None):
        super().__init__(model_config, model_type, device=device, unet_model=comfy.ldm.omnigen.omnigen2.OmniGen2Transformer2DModel)
--- a/comfy/model_detection.py
+++ b/comfy/model_detection.py
@ -655,6 +655,11 @@ def detect_unet_config(state_dict, key_prefix, metadata=None):
        dit_config["num_visual_blocks"] = count_blocks(state_dict_keys, '{}visual_transformer_blocks.'.format(key_prefix) + '{}.')
        return dit_config

+    if '{}encoder.lyric_encoder.layers.0.input_layernorm.weight'.format(key_prefix) in state_dict_keys:
+        dit_config = {}
+        dit_config["audio_model"] = "ace1.5"
+        return dit_config
+
    if '{}input_blocks.0.0.weight'.format(key_prefix) not in state_dict_keys:
        return None

--- a/comfy/model_management.py
+++ b/comfy/model_management.py
@ -19,7 +19,8 @@
 import psutil
 import logging
 from enum import Enum
-from comfy.cli_args import args, PerformanceFeature
+from comfy.cli_args import args, PerformanceFeature, enables_dynamic_vram
+import threading
 import torch
 import sys
 import platform
@ -650,7 +651,7 @@ def free_memory(memory_required, device, keep_loaded=[], for_dynamic=False, ram_
                soft_empty_cache()
    return unloaded_models

-def load_models_gpu(models, memory_required=0, force_patch_weights=False, minimum_memory_required=None, force_full_load=False):
+def load_models_gpu_orig(models, memory_required=0, force_patch_weights=False, minimum_memory_required=None, force_full_load=False):
    cleanup_models_gc()
    global vram_state

@ -746,6 +747,26 @@ def load_models_gpu(models, memory_required=0, force_patch_weights=False, minimu
        current_loaded_models.insert(0, loaded_model)
    return

+def load_models_gpu_thread(models, memory_required, force_patch_weights, minimum_memory_required, force_full_load):
+    with torch.inference_mode():
+        load_models_gpu_orig(models, memory_required, force_patch_weights, minimum_memory_required, force_full_load)
+        soft_empty_cache()
+
+def load_models_gpu(models, memory_required=0, force_patch_weights=False, minimum_memory_required=None, force_full_load=False):
+    #Deliberately load models outside of the Aimdo mempool so they can be retained accross
+    #nodes. Use a dummy thread to do it as pytorch documents that mempool contexts are
+    #thread local. So exploit that to escape context
+    if enables_dynamic_vram():
+        t = threading.Thread(
+            target=load_models_gpu_thread,
+            args=(models, memory_required, force_patch_weights, minimum_memory_required, force_full_load)
+        )
+        t.start()
+        t.join()
+    else:
+        load_models_gpu_orig(models, memory_required=memory_required, force_patch_weights=force_patch_weights,
+                             minimum_memory_required=minimum_memory_required, force_full_load=force_full_load)
+
 def load_model_gpu(model):
    return load_models_gpu([model])

@ -1112,11 +1133,11 @@ def get_cast_buffer(offload_stream, device, size, ref):
            return None
        if cast_buffer is not None and cast_buffer.numel() > 50 * (1024 ** 2):
            #I want my wrongly sized 50MB+ of VRAM back from the caching allocator right now
-            torch.cuda.synchronize()
+            synchronize()
            del STREAM_CAST_BUFFERS[offload_stream]
            del cast_buffer
            #FIXME: This doesn't work in Aimdo because mempool cant clear cache
-            torch.cuda.empty_cache()
+            soft_empty_cache()
        with wf_context:
            cast_buffer = torch.empty((size), dtype=torch.int8, device=device)
            STREAM_CAST_BUFFERS[offload_stream] = cast_buffer
@ -1132,9 +1153,7 @@ def reset_cast_buffers():
    for offload_stream in STREAM_CAST_BUFFERS:
        offload_stream.synchronize()
    STREAM_CAST_BUFFERS.clear()
-    if comfy.memory_management.aimdo_allocator is None:
-        #Pytorch 2.7 and earlier crashes if you try and empty_cache when mempools exist
-        torch.cuda.empty_cache()
+    soft_empty_cache()

 def get_offload_stream(device):
    stream_counter = stream_counters.get(device, 0)
@ -1284,7 +1303,7 @@ def discard_cuda_async_error():
        a = torch.tensor([1], dtype=torch.uint8, device=get_torch_device())
        b = torch.tensor([1], dtype=torch.uint8, device=get_torch_device())
        _ = a + b
-        torch.cuda.synchronize()
+        synchronize()
    except torch.AcceleratorError:
        #Dump it! We already know about it from the synchronous return
        pass
@ -1688,6 +1707,12 @@ def lora_compute_dtype(device):
    LORA_COMPUTE_DTYPES[device] = dtype
    return dtype

+def synchronize():
+    if is_intel_xpu():
+        torch.xpu.synchronize()
+    elif torch.cuda.is_available():
+        torch.cuda.synchronize()
+
 def soft_empty_cache(force=False):
    global cpu_state
    if cpu_state == CPUState.MPS:
@ -1713,9 +1738,6 @@ def debug_memory_summary():
        return torch.cuda.memory.memory_summary()
    return ""

-#TODO: might be cleaner to put this somewhere else
-import threading
-
 class InterruptProcessingException(Exception):
    pass

--- a/comfy/model_patcher.py
+++ b/comfy/model_patcher.py
@ -1597,7 +1597,7 @@ class ModelPatcherDynamic(ModelPatcher):

        if unpatch_weights:
            self.partially_unload_ram(1e32)
-            self.partially_unload(None)
+            self.partially_unload(None, 1e32)

    def partially_load(self, device_to, extra_memory=0, force_patch_weights=False):
        assert not force_patch_weights #See above
--- a/comfy/sd.py
+++ b/comfy/sd.py
@ -59,6 +59,7 @@ import comfy.text_encoders.kandinsky5
 import comfy.text_encoders.jina_clip_2
 import comfy.text_encoders.newbie
 import comfy.text_encoders.anima
+import comfy.text_encoders.ace15

 import comfy.model_patcher
 import comfy.lora
@ -452,6 +453,8 @@ class VAE:
        self.extra_1d_channel = None
        self.crop_input = True

+        self.audio_sample_rate = 44100
+
        if config is None:
            if "decoder.mid.block_1.mix_factor" in sd:
                encoder_config = {'double_z': True, 'z_channels': 4, 'resolution': 256, 'in_channels': 3, 'out_ch': 3, 'ch': 128, 'ch_mult': [1, 2, 4, 4], 'num_res_blocks': 2, 'attn_resolutions': [], 'dropout': 0.0}
@ -549,14 +552,25 @@ class VAE:
                                                                    encoder_config={'target': "comfy.ldm.modules.diffusionmodules.model.Encoder", 'params': ddconfig},
                                                                    decoder_config={'target': "comfy.ldm.modules.diffusionmodules.model.Decoder", 'params': ddconfig})
            elif "decoder.layers.1.layers.0.beta" in sd:
-                self.first_stage_model = AudioOobleckVAE()
+                config = {}
+                param_key = None
+                if "decoder.layers.2.layers.1.weight_v" in sd:
+                    param_key = "decoder.layers.2.layers.1.weight_v"
+                if "decoder.layers.2.layers.1.parametrizations.weight.original1" in sd:
+                    param_key = "decoder.layers.2.layers.1.parametrizations.weight.original1"
+                if param_key is not None:
+                    if sd[param_key].shape[-1] == 12:
+                        config["strides"] = [2, 4, 4, 6, 10]
+                        self.audio_sample_rate = 48000
+
+                self.first_stage_model = AudioOobleckVAE(**config)
                self.memory_used_encode = lambda shape, dtype: (1000 * shape[2]) * model_management.dtype_size(dtype)
                self.memory_used_decode = lambda shape, dtype: (1000 * shape[2] * 2048) * model_management.dtype_size(dtype)
                self.latent_channels = 64
                self.output_channels = 2
                self.pad_channel_value = "replicate"
                self.upscale_ratio = 2048
-                self.downscale_ratio =  2048
+                self.downscale_ratio = 2048
                self.latent_dim = 1
                self.process_output = lambda audio: audio
                self.process_input = lambda audio: audio
@ -1427,6 +1441,9 @@ def load_text_encoder_state_dicts(state_dicts=[], embedding_directory=None, clip
                clip_data_jina = clip_data[0]
            tokenizer_data["gemma_spiece_model"] = clip_data_gemma.get("spiece_model", None)
            tokenizer_data["jina_spiece_model"] = clip_data_jina.get("spiece_model", None)
+        elif clip_type == CLIPType.ACE:
+            clip_target.clip = comfy.text_encoders.ace15.te(**llama_detect(clip_data))
+            clip_target.tokenizer = comfy.text_encoders.ace15.ACE15Tokenizer
        else:
            clip_target.clip = sdxl_clip.SDXLClipModel
            clip_target.tokenizer = sdxl_clip.SDXLTokenizer
--- a/comfy/sd1_clip.py
+++ b/comfy/sd1_clip.py
@ -155,6 +155,8 @@ class SDClipModel(torch.nn.Module, ClipTokenWeightEncoder):
        self.execution_device = options.get("execution_device", self.execution_device)
        if isinstance(self.layer, list) or self.layer == "all":
            pass
+        elif isinstance(layer_idx, list):
+            self.layer = layer_idx
        elif layer_idx is None or abs(layer_idx) > self.num_layers:
            self.layer = "last"
        else:
--- a/comfy/supported_models.py
+++ b/comfy/supported_models.py
@ -24,6 +24,7 @@ import comfy.text_encoders.hunyuan_image
 import comfy.text_encoders.kandinsky5
 import comfy.text_encoders.z_image
 import comfy.text_encoders.anima
+import comfy.text_encoders.ace15

 from . import supported_models_base
 from . import latent_formats
@ -1596,6 +1597,38 @@ class Kandinsky5Image(Kandinsky5):
        return supported_models_base.ClipTarget(comfy.text_encoders.kandinsky5.Kandinsky5TokenizerImage, comfy.text_encoders.kandinsky5.te(**hunyuan_detect))


-models = [LotusD, Stable_Zero123, SD15_instructpix2pix, SD15, SD20, SD21UnclipL, SD21UnclipH, SDXL_instructpix2pix, SDXLRefiner, SDXL, SSD1B, KOALA_700M, KOALA_1B, Segmind_Vega, SD_X4Upscaler, Stable_Cascade_C, Stable_Cascade_B, SV3D_u, SV3D_p, SD3, StableAudio, AuraFlow, PixArtAlpha, PixArtSigma, HunyuanDiT, HunyuanDiT1, FluxInpaint, Flux, FluxSchnell, GenmoMochi, LTXV, LTXAV, HunyuanVideo15_SR_Distilled, HunyuanVideo15, HunyuanImage21Refiner, HunyuanImage21, HunyuanVideoSkyreelsI2V, HunyuanVideoI2V, HunyuanVideo, CosmosT2V, CosmosI2V, CosmosT2IPredict2, CosmosI2VPredict2, ZImage, Lumina2, WAN22_T2V, WAN21_T2V, WAN21_I2V, WAN21_FunControl2V, WAN21_Vace, WAN21_Camera, WAN22_Camera, WAN22_S2V, WAN21_HuMo, WAN22_Animate, Hunyuan3Dv2mini, Hunyuan3Dv2, Hunyuan3Dv2_1, HiDream, Chroma, ChromaRadiance, ACEStep, Omnigen2, QwenImage, Flux2, Kandinsky5Image, Kandinsky5, Anima]
+class ACEStep15(supported_models_base.BASE):
+    unet_config = {
+        "audio_model": "ace1.5",
+    }
+
+    unet_extra_config = {
+    }
+
+    sampling_settings = {
+        "multiplier": 1.0,
+        "shift": 3.0,
+    }
+
+    latent_format = comfy.latent_formats.ACEAudio15
+
+    memory_usage_factor = 4.7
+
+    supported_inference_dtypes = [torch.bfloat16, torch.float32]
+
+    vae_key_prefix = ["vae."]
+    text_encoder_key_prefix = ["text_encoders."]
+
+    def get_model(self, state_dict, prefix="", device=None):
+        out = model_base.ACEStep15(self, device=device)
+        return out
+
+    def clip_target(self, state_dict={}):
+        pref = self.text_encoder_key_prefix[0]
+        hunyuan_detect = comfy.text_encoders.hunyuan_video.llama_detect(state_dict, "{}qwen3_2b.transformer.".format(pref))
+        return supported_models_base.ClipTarget(comfy.text_encoders.ace15.ACE15Tokenizer, comfy.text_encoders.ace15.te(**hunyuan_detect))
+
+
+models = [LotusD, Stable_Zero123, SD15_instructpix2pix, SD15, SD20, SD21UnclipL, SD21UnclipH, SDXL_instructpix2pix, SDXLRefiner, SDXL, SSD1B, KOALA_700M, KOALA_1B, Segmind_Vega, SD_X4Upscaler, Stable_Cascade_C, Stable_Cascade_B, SV3D_u, SV3D_p, SD3, StableAudio, AuraFlow, PixArtAlpha, PixArtSigma, HunyuanDiT, HunyuanDiT1, FluxInpaint, Flux, FluxSchnell, GenmoMochi, LTXV, LTXAV, HunyuanVideo15_SR_Distilled, HunyuanVideo15, HunyuanImage21Refiner, HunyuanImage21, HunyuanVideoSkyreelsI2V, HunyuanVideoI2V, HunyuanVideo, CosmosT2V, CosmosI2V, CosmosT2IPredict2, CosmosI2VPredict2, ZImage, Lumina2, WAN22_T2V, WAN21_T2V, WAN21_I2V, WAN21_FunControl2V, WAN21_Vace, WAN21_Camera, WAN22_Camera, WAN22_S2V, WAN21_HuMo, WAN22_Animate, Hunyuan3Dv2mini, Hunyuan3Dv2, Hunyuan3Dv2_1, HiDream, Chroma, ChromaRadiance, ACEStep, ACEStep15, Omnigen2, QwenImage, Flux2, Kandinsky5Image, Kandinsky5, Anima]

 models += [SVD_img2vid]
--- a/comfy/text_encoders/ace15.py
+++ b/comfy/text_encoders/ace15.py
@ -0,0 +1,222 @@
+from .anima import Qwen3Tokenizer
+import comfy.text_encoders.llama
+from comfy import sd1_clip
+import torch
+import math
+import comfy.utils
+
+
+def sample_manual_loop_no_classes(
+    model,
+    ids=None,
+    paddings=[],
+    execution_dtype=None,
+    cfg_scale: float = 2.0,
+    temperature: float = 0.85,
+    top_p: float = 0.9,
+    top_k: int = None,
+    seed: int = 1,
+    min_tokens: int = 1,
+    max_new_tokens: int = 2048,
+    audio_start_id: int = 151669,  # The cutoff ID for audio codes
+    eos_token_id: int = 151645,
+):
+    device = model.execution_device
+
+    if execution_dtype is None:
+        if comfy.model_management.should_use_bf16(device):
+            execution_dtype = torch.bfloat16
+        else:
+            execution_dtype = torch.float32
+
+    embeds, attention_mask, num_tokens, embeds_info = model.process_tokens(ids, device)
+    for i, t in enumerate(paddings):
+        attention_mask[i, :t] = 0
+        attention_mask[i, t:] = 1
+
+    output_audio_codes = []
+    past_key_values = []
+    generator = torch.Generator(device=device)
+    generator.manual_seed(seed)
+    model_config = model.transformer.model.config
+
+    for x in range(model_config.num_hidden_layers):
+        past_key_values.append((torch.empty([embeds.shape[0], model_config.num_key_value_heads, embeds.shape[1] + min_tokens, model_config.head_dim], device=device, dtype=execution_dtype), torch.empty([embeds.shape[0], model_config.num_key_value_heads, embeds.shape[1] + min_tokens, model_config.head_dim], device=device, dtype=execution_dtype), 0))
+
+    progress_bar = comfy.utils.ProgressBar(max_new_tokens)
+
+    for step in range(max_new_tokens):
+        outputs = model.transformer(None, attention_mask, embeds=embeds.to(execution_dtype), num_tokens=num_tokens, intermediate_output=None, dtype=execution_dtype, embeds_info=embeds_info, past_key_values=past_key_values)
+        next_token_logits = model.transformer.logits(outputs[0])[:, -1]
+        past_key_values = outputs[2]
+
+        cond_logits = next_token_logits[0:1]
+        uncond_logits = next_token_logits[1:2]
+        cfg_logits = uncond_logits + cfg_scale * (cond_logits - uncond_logits)
+
+        if eos_token_id is not None and eos_token_id < audio_start_id and min_tokens < step:
+            eos_score = cfg_logits[:, eos_token_id].clone()
+
+        # Only generate audio tokens
+        cfg_logits[:, :audio_start_id] = float('-inf')
+
+        if eos_token_id is not None and eos_token_id < audio_start_id and min_tokens < step:
+            cfg_logits[:, eos_token_id] = eos_score
+
+        if top_k is not None and top_k > 0:
+            top_k_vals, _ = torch.topk(cfg_logits, top_k)
+            min_val = top_k_vals[..., -1, None]
+            cfg_logits[cfg_logits < min_val] = float('-inf')
+
+        if top_p is not None and top_p < 1.0:
+            sorted_logits, sorted_indices = torch.sort(cfg_logits, descending=True)
+            cumulative_probs = torch.cumsum(torch.softmax(sorted_logits, dim=-1), dim=-1)
+            sorted_indices_to_remove = cumulative_probs > top_p
+            sorted_indices_to_remove[..., 1:] = sorted_indices_to_remove[..., :-1].clone()
+            sorted_indices_to_remove[..., 0] = 0
+            indices_to_remove = sorted_indices_to_remove.scatter(1, sorted_indices, sorted_indices_to_remove)
+            cfg_logits[indices_to_remove] = float('-inf')
+
+        if temperature > 0:
+            cfg_logits = cfg_logits / temperature
+            next_token = torch.multinomial(torch.softmax(cfg_logits, dim=-1), num_samples=1, generator=generator).squeeze(1)
+        else:
+            next_token = torch.argmax(cfg_logits, dim=-1)
+
+        token = next_token.item()
+
+        if token == eos_token_id:
+            break
+
+        embed, _, _, _ = model.process_tokens([[token]], device)
+        embeds = embed.repeat(2, 1, 1)
+        attention_mask = torch.cat([attention_mask, torch.ones((2, 1), device=device, dtype=attention_mask.dtype)], dim=1)
+
+        output_audio_codes.append(token - audio_start_id)
+        progress_bar.update_absolute(step)
+
+    return output_audio_codes
+
+
+def generate_audio_codes(model, positive, negative, min_tokens=1, max_tokens=1024, seed=0):
+    cfg_scale = 2.0
+
+    positive = [[token for token, _ in inner_list] for inner_list in positive]
+    negative = [[token for token, _ in inner_list] for inner_list in negative]
+    positive = positive[0]
+    negative = negative[0]
+
+    neg_pad = 0
+    if len(negative) < len(positive):
+        neg_pad = (len(positive) - len(negative))
+        negative = [model.special_tokens["pad"]] * neg_pad + negative
+
+    pos_pad = 0
+    if len(negative) > len(positive):
+        pos_pad = (len(negative) - len(positive))
+        positive = [model.special_tokens["pad"]] * pos_pad + positive
+
+    paddings = [pos_pad, neg_pad]
+    return sample_manual_loop_no_classes(model, [positive, negative], paddings, cfg_scale=cfg_scale, seed=seed, min_tokens=min_tokens, max_new_tokens=max_tokens)
+
+
+class ACE15Tokenizer(sd1_clip.SD1Tokenizer):
+    def __init__(self, embedding_directory=None, tokenizer_data={}):
+        super().__init__(embedding_directory=embedding_directory, tokenizer_data=tokenizer_data, name="qwen3_06b", tokenizer=Qwen3Tokenizer)
+
+    def tokenize_with_weights(self, text, return_word_ids=False, **kwargs):
+        out = {}
+        lyrics = kwargs.get("lyrics", "")
+        bpm = kwargs.get("bpm", 120)
+        duration = kwargs.get("duration", 120)
+        keyscale = kwargs.get("keyscale", "C major")
+        timesignature = kwargs.get("timesignature", 2)
+        language = kwargs.get("language", "en")
+        seed = kwargs.get("seed", 0)
+
+        duration = math.ceil(duration)
+        meta_lm = 'bpm: {}\nduration: {}\nkeyscale: {}\ntimesignature: {}'.format(bpm, duration, keyscale, timesignature)
+        lm_template = "<|im_start|>system\n# Instruction\nGenerate audio semantic tokens based on the given conditions:\n\n<|im_end|>\n<|im_start|>user\n# Caption\n{}\n{}\n<|im_end|>\n<|im_start|>assistant\n<think>\n{}\n</think>\n\n<|im_end|>\n"
+
+        meta_cap = '- bpm: {}\n- timesignature: {}\n- keyscale: {}\n- duration: {}\n'.format(bpm, timesignature, keyscale, duration)
+        out["lm_prompt"] = self.qwen3_06b.tokenize_with_weights(lm_template.format(text, lyrics, meta_lm), disable_weights=True)
+        out["lm_prompt_negative"] = self.qwen3_06b.tokenize_with_weights(lm_template.format(text, lyrics, ""), disable_weights=True)
+
+        out["lyrics"] = self.qwen3_06b.tokenize_with_weights("# Languages\n{}\n\n# Lyric{}<|endoftext|><|endoftext|>".format(language, lyrics), return_word_ids, disable_weights=True, **kwargs)
+        out["qwen3_06b"] = self.qwen3_06b.tokenize_with_weights("# Instruction\nGenerate audio semantic tokens based on the given conditions:\n\n# Caption\n{}# Metas\n{}<|endoftext|>\n<|endoftext|>".format(text, meta_cap), return_word_ids, **kwargs)
+        out["lm_metadata"] = {"min_tokens": duration * 5, "seed": seed}
+        return out
+
+
+class Qwen3_06BModel(sd1_clip.SDClipModel):
+    def __init__(self, device="cpu", layer="last", layer_idx=None, dtype=None, attention_mask=True, model_options={}):
+        super().__init__(device=device, layer=layer, layer_idx=layer_idx, textmodel_json_config={}, dtype=dtype, special_tokens={"pad": 151643}, layer_norm_hidden_state=False, model_class=comfy.text_encoders.llama.Qwen3_06B_ACE15, enable_attention_masks=attention_mask, return_attention_masks=attention_mask, model_options=model_options)
+
+class Qwen3_2B_ACE15(sd1_clip.SDClipModel):
+    def __init__(self, device="cpu", layer="last", layer_idx=None, dtype=None, attention_mask=True, model_options={}):
+        llama_quantization_metadata = model_options.get("llama_quantization_metadata", None)
+        if llama_quantization_metadata is not None:
+            model_options = model_options.copy()
+            model_options["quantization_metadata"] = llama_quantization_metadata
+
+        super().__init__(device=device, layer=layer, layer_idx=layer_idx, textmodel_json_config={}, dtype=dtype, special_tokens={"pad": 151643}, layer_norm_hidden_state=False, model_class=comfy.text_encoders.llama.Qwen3_2B_ACE15_lm, enable_attention_masks=attention_mask, return_attention_masks=attention_mask, model_options=model_options)
+
+class ACE15TEModel(torch.nn.Module):
+    def __init__(self, device="cpu", dtype=None, dtype_llama=None, model_options={}):
+        super().__init__()
+        if dtype_llama is None:
+            dtype_llama = dtype
+
+        self.qwen3_06b = Qwen3_06BModel(device=device, dtype=dtype, model_options=model_options)
+        self.qwen3_2b = Qwen3_2B_ACE15(device=device, dtype=dtype_llama, model_options=model_options)
+        self.dtypes = set([dtype, dtype_llama])
+
+    def encode_token_weights(self, token_weight_pairs):
+        token_weight_pairs_base = token_weight_pairs["qwen3_06b"]
+        token_weight_pairs_lyrics = token_weight_pairs["lyrics"]
+
+        self.qwen3_06b.set_clip_options({"layer": None})
+        base_out, _, extra = self.qwen3_06b.encode_token_weights(token_weight_pairs_base)
+        self.qwen3_06b.set_clip_options({"layer": [0]})
+        lyrics_embeds, _, extra_l = self.qwen3_06b.encode_token_weights(token_weight_pairs_lyrics)
+
+        lm_metadata = token_weight_pairs["lm_metadata"]
+        audio_codes = generate_audio_codes(self.qwen3_2b, token_weight_pairs["lm_prompt"], token_weight_pairs["lm_prompt_negative"], min_tokens=lm_metadata["min_tokens"], max_tokens=lm_metadata["min_tokens"], seed=lm_metadata["seed"])
+
+        return base_out, None, {"conditioning_lyrics": lyrics_embeds[:, 0], "audio_codes": [audio_codes]}
+
+    def set_clip_options(self, options):
+        self.qwen3_06b.set_clip_options(options)
+        self.qwen3_2b.set_clip_options(options)
+
+    def reset_clip_options(self):
+        self.qwen3_06b.reset_clip_options()
+        self.qwen3_2b.reset_clip_options()
+
+    def load_sd(self, sd):
+        if "model.layers.0.post_attention_layernorm.weight" in sd:
+            shape = sd["model.layers.0.post_attention_layernorm.weight"].shape
+            if shape[0] == 1024:
+                return self.qwen3_06b.load_sd(sd)
+            else:
+                return self.qwen3_2b.load_sd(sd)
+
+    def memory_estimation_function(self, token_weight_pairs, device=None):
+        lm_metadata = token_weight_pairs["lm_metadata"]
+        constant = 0.4375
+        if comfy.model_management.should_use_bf16(device):
+            constant *= 0.5
+
+        token_weight_pairs = token_weight_pairs.get("lm_prompt", [])
+        num_tokens = sum(map(lambda a: len(a), token_weight_pairs))
+        num_tokens += lm_metadata['min_tokens']
+        return num_tokens * constant * 1024 * 1024
+
+def te(dtype_llama=None, llama_quantization_metadata=None):
+    class ACE15TEModel_(ACE15TEModel):
+        def __init__(self, device="cpu", dtype=None, model_options={}):
+            if llama_quantization_metadata is not None:
+                model_options = model_options.copy()
+                model_options["llama_quantization_metadata"] = llama_quantization_metadata
+            super().__init__(device=device, dtype_llama=dtype_llama, dtype=dtype, model_options=model_options)
+    return ACE15TEModel_
--- a/comfy/text_encoders/llama.py
+++ b/comfy/text_encoders/llama.py
@ -103,6 +103,52 @@ class Qwen3_06BConfig:
    final_norm: bool = True
    lm_head: bool = False

+@dataclass
+class Qwen3_06B_ACE15_Config:
+    vocab_size: int = 151669
+    hidden_size: int = 1024
+    intermediate_size: int = 3072
+    num_hidden_layers: int = 28
+    num_attention_heads: int = 16
+    num_key_value_heads: int = 8
+    max_position_embeddings: int = 32768
+    rms_norm_eps: float = 1e-6
+    rope_theta: float = 1000000.0
+    transformer_type: str = "llama"
+    head_dim = 128
+    rms_norm_add = False
+    mlp_activation = "silu"
+    qkv_bias = False
+    rope_dims = None
+    q_norm = "gemma3"
+    k_norm = "gemma3"
+    rope_scale = None
+    final_norm: bool = True
+    lm_head: bool = False
+
+@dataclass
+class Qwen3_2B_ACE15_lm_Config:
+    vocab_size: int = 217204
+    hidden_size: int = 2048
+    intermediate_size: int = 6144
+    num_hidden_layers: int = 28
+    num_attention_heads: int = 16
+    num_key_value_heads: int = 8
+    max_position_embeddings: int = 40960
+    rms_norm_eps: float = 1e-6
+    rope_theta: float = 1000000.0
+    transformer_type: str = "llama"
+    head_dim = 128
+    rms_norm_add = False
+    mlp_activation = "silu"
+    qkv_bias = False
+    rope_dims = None
+    q_norm = "gemma3"
+    k_norm = "gemma3"
+    rope_scale = None
+    final_norm: bool = True
+    lm_head: bool = False
+
@dataclass
 class Qwen3_4BConfig:
    vocab_size: int = 151936
@ -729,6 +775,27 @@ class Qwen3_06B(BaseLlama, torch.nn.Module):
        self.model = Llama2_(config, device=device, dtype=dtype, ops=operations)
        self.dtype = dtype

+class Qwen3_06B_ACE15(BaseLlama, torch.nn.Module):
+    def __init__(self, config_dict, dtype, device, operations):
+        super().__init__()
+        config = Qwen3_06B_ACE15_Config(**config_dict)
+        self.num_layers = config.num_hidden_layers
+
+        self.model = Llama2_(config, device=device, dtype=dtype, ops=operations)
+        self.dtype = dtype
+
+class Qwen3_2B_ACE15_lm(BaseLlama, torch.nn.Module):
+    def __init__(self, config_dict, dtype, device, operations):
+        super().__init__()
+        config = Qwen3_2B_ACE15_lm_Config(**config_dict)
+        self.num_layers = config.num_hidden_layers
+
+        self.model = Llama2_(config, device=device, dtype=dtype, ops=operations)
+        self.dtype = dtype
+
+    def logits(self, x):
+        return torch.nn.functional.linear(x[:, -1:], self.model.embed_tokens.weight.to(x), None)
+
 class Qwen3_4B(BaseLlama, torch.nn.Module):
    def __init__(self, config_dict, dtype, device, operations):
        super().__init__()
--- a/comfy_api_nodes/apis/hitpaw.py
+++ b/comfy_api_nodes/apis/hitpaw.py
@ -0,0 +1,51 @@
+from typing import TypedDict
+
+from pydantic import BaseModel, Field
+
+
+class InputVideoModel(TypedDict):
+    model: str
+    resolution: str
+
+
+class ImageEnhanceTaskCreateRequest(BaseModel):
+    model_name: str = Field(...)
+    img_url: str = Field(...)
+    extension: str = Field(".png")
+    exif: bool = Field(False)
+    DPI: int | None = Field(None)
+
+
+class VideoEnhanceTaskCreateRequest(BaseModel):
+    video_url: str = Field(...)
+    extension: str = Field(".mp4")
+    model_name: str | None = Field(...)
+    resolution: list[int] = Field(..., description="Target resolution [width, height]")
+    original_resolution: list[int] = Field(..., description="Original video resolution [width, height]")
+
+
+class TaskCreateDataResponse(BaseModel):
+    job_id: str = Field(...)
+    consume_coins: int | None = Field(None)
+
+
+class TaskStatusPollRequest(BaseModel):
+    job_id: str = Field(...)
+
+
+class TaskCreateResponse(BaseModel):
+    code: int = Field(...)
+    message: str = Field(...)
+    data: TaskCreateDataResponse | None = Field(None)
+
+
+class TaskStatusDataResponse(BaseModel):
+    job_id: str = Field(...)
+    status: str = Field(...)
+    res_url: str = Field("")
+
+
+class TaskStatusResponse(BaseModel):
+    code: int = Field(...)
+    message: str = Field(...)
+    data: TaskStatusDataResponse = Field(...)
--- a/comfy_api_nodes/nodes_hitpaw.py
+++ b/comfy_api_nodes/nodes_hitpaw.py
@ -0,0 +1,342 @@
+import math
+
+from typing_extensions import override
+
+from comfy_api.latest import IO, ComfyExtension, Input
+from comfy_api_nodes.apis.hitpaw import (
+    ImageEnhanceTaskCreateRequest,
+    InputVideoModel,
+    TaskCreateDataResponse,
+    TaskCreateResponse,
+    TaskStatusPollRequest,
+    TaskStatusResponse,
+    VideoEnhanceTaskCreateRequest,
+)
+from comfy_api_nodes.util import (
+    ApiEndpoint,
+    download_url_to_image_tensor,
+    download_url_to_video_output,
+    downscale_image_tensor,
+    get_image_dimensions,
+    poll_op,
+    sync_op,
+    upload_image_to_comfyapi,
+    upload_video_to_comfyapi,
+    validate_video_duration,
+)
+
+VIDEO_MODELS_MODELS_MAP = {
+    "Portrait Restore Model (1x)": "portrait_restore_1x",
+    "Portrait Restore Model (2x)": "portrait_restore_2x",
+    "General Restore Model (1x)": "general_restore_1x",
+    "General Restore Model (2x)": "general_restore_2x",
+    "General Restore Model (4x)": "general_restore_4x",
+    "Ultra HD Model (2x)": "ultrahd_restore_2x",
+    "Generative Model (1x)": "generative_1x",
+}
+
+# Resolution name to target dimension (shorter side) in pixels
+RESOLUTION_TARGET_MAP = {
+    "720p": 720,
+    "1080p": 1080,
+    "2K/QHD": 1440,
+    "4K/UHD": 2160,
+    "8K": 4320,
+}
+
+# Square (1:1) resolutions use standard square dimensions
+RESOLUTION_SQUARE_MAP = {
+    "720p": 720,
+    "1080p": 1080,
+    "2K/QHD": 1440,
+    "4K/UHD": 2048,  # DCI 4K square
+    "8K": 4096,  # DCI 8K square
+}
+
+# Models with limited resolution support (no 8K)
+LIMITED_RESOLUTION_MODELS = {"Generative Model (1x)"}
+
+# Resolution options for different model types
+RESOLUTIONS_LIMITED = ["original", "720p", "1080p", "2K/QHD", "4K/UHD"]
+RESOLUTIONS_FULL = ["original", "720p", "1080p", "2K/QHD", "4K/UHD", "8K"]
+
+# Maximum output resolution in pixels
+MAX_PIXELS_GENERATIVE = 32_000_000
+MAX_MP_GENERATIVE = MAX_PIXELS_GENERATIVE // 1_000_000
+
+
+class HitPawGeneralImageEnhance(IO.ComfyNode):
+    @classmethod
+    def define_schema(cls):
+        return IO.Schema(
+            node_id="HitPawGeneralImageEnhance",
+            display_name="HitPaw General Image Enhance",
+            category="api node/image/HitPaw",
+            description="Upscale low-resolution images to super-resolution, eliminate artifacts and noise. "
+            f"Maximum output: {MAX_MP_GENERATIVE} megapixels.",
+            inputs=[
+                IO.Combo.Input("model", options=["generative_portrait", "generative"]),
+                IO.Image.Input("image"),
+                IO.Combo.Input("upscale_factor", options=[1, 2, 4]),
+                IO.Boolean.Input(
+                    "auto_downscale",
+                    default=False,
+                    tooltip="Automatically downscale input image if output would exceed the limit.",
+                ),
+            ],
+            outputs=[
+                IO.Image.Output(),
+            ],
+            hidden=[
+                IO.Hidden.auth_token_comfy_org,
+                IO.Hidden.api_key_comfy_org,
+                IO.Hidden.unique_id,
+            ],
+            is_api_node=True,
+            price_badge=IO.PriceBadge(
+                depends_on=IO.PriceBadgeDepends(widgets=["model"]),
+                expr="""
+                (
+                  $prices := {
+                    "generative_portrait": {"min": 0.02, "max": 0.06},
+                    "generative": {"min": 0.05, "max": 0.15}
+                  };
+                  $price := $lookup($prices, widgets.model);
+                  {
+                    "type": "range_usd",
+                    "min_usd": $price.min,
+                    "max_usd": $price.max
+                  }
+                )
+                """,
+            ),
+        )
+
+    @classmethod
+    async def execute(
+        cls,
+        model: str,
+        image: Input.Image,
+        upscale_factor: int,
+        auto_downscale: bool,
+    ) -> IO.NodeOutput:
+        height, width = get_image_dimensions(image)
+        requested_scale = upscale_factor
+        output_pixels = height * width * requested_scale * requested_scale
+        if output_pixels > MAX_PIXELS_GENERATIVE:
+            if auto_downscale:
+                input_pixels = width * height
+                scale = 1
+                max_input_pixels = MAX_PIXELS_GENERATIVE
+
+                for candidate in [4, 2, 1]:
+                    if candidate > requested_scale:
+                        continue
+                    scale_output_pixels = input_pixels * candidate * candidate
+                    if scale_output_pixels <= MAX_PIXELS_GENERATIVE:
+                        scale = candidate
+                        max_input_pixels = None
+                        break
+                    # Check if we can downscale input by at most 2x to fit
+                    downscale_ratio = math.sqrt(scale_output_pixels / MAX_PIXELS_GENERATIVE)
+                    if downscale_ratio <= 2.0:
+                        scale = candidate
+                        max_input_pixels = MAX_PIXELS_GENERATIVE // (candidate * candidate)
+                        break
+
+                if max_input_pixels is not None:
+                    image = downscale_image_tensor(image, total_pixels=max_input_pixels)
+                upscale_factor = scale
+            else:
+                output_width = width * requested_scale
+                output_height = height * requested_scale
+                raise ValueError(
+                    f"Output size ({output_width}x{output_height} = {output_pixels:,} pixels) "
+                    f"exceeds maximum allowed size of {MAX_PIXELS_GENERATIVE:,} pixels ({MAX_MP_GENERATIVE}MP). "
+                    f"Enable auto_downscale or use a smaller input image or a lower upscale factor."
+                )
+
+        initial_res = await sync_op(
+            cls,
+            ApiEndpoint(path="/proxy/hitpaw/api/photo-enhancer", method="POST"),
+            response_model=TaskCreateResponse,
+            data=ImageEnhanceTaskCreateRequest(
+                model_name=f"{model}_{upscale_factor}x",
+                img_url=await upload_image_to_comfyapi(cls, image, total_pixels=None),
+            ),
+            wait_label="Creating task",
+            final_label_on_success="Task created",
+        )
+        if initial_res.code != 200:
+            raise ValueError(f"Task creation failed with code {initial_res.code}: {initial_res.message}")
+        request_price = initial_res.data.consume_coins / 1000
+        final_response = await poll_op(
+            cls,
+            ApiEndpoint(path="/proxy/hitpaw/api/task-status", method="POST"),
+            data=TaskCreateDataResponse(job_id=initial_res.data.job_id),
+            response_model=TaskStatusResponse,
+            status_extractor=lambda x: x.data.status,
+            price_extractor=lambda x: request_price,
+            poll_interval=10.0,
+            max_poll_attempts=480,
+        )
+        return IO.NodeOutput(await download_url_to_image_tensor(final_response.data.res_url))
+
+
+class HitPawVideoEnhance(IO.ComfyNode):
+    @classmethod
+    def define_schema(cls):
+        model_options = []
+        for model_name in VIDEO_MODELS_MODELS_MAP:
+            if model_name in LIMITED_RESOLUTION_MODELS:
+                resolutions = RESOLUTIONS_LIMITED
+            else:
+                resolutions = RESOLUTIONS_FULL
+            model_options.append(
+                IO.DynamicCombo.Option(
+                    model_name,
+                    [IO.Combo.Input("resolution", options=resolutions)],
+                )
+            )
+
+        return IO.Schema(
+            node_id="HitPawVideoEnhance",
+            display_name="HitPaw Video Enhance",
+            category="api node/video/HitPaw",
+            description="Upscale low-resolution videos to high resolution, eliminate artifacts and noise. "
+            "Prices shown are per second of video.",
+            inputs=[
+                IO.DynamicCombo.Input("model", options=model_options),
+                IO.Video.Input("video"),
+            ],
+            outputs=[
+                IO.Video.Output(),
+            ],
+            hidden=[
+                IO.Hidden.auth_token_comfy_org,
+                IO.Hidden.api_key_comfy_org,
+                IO.Hidden.unique_id,
+            ],
+            is_api_node=True,
+            price_badge=IO.PriceBadge(
+                depends_on=IO.PriceBadgeDepends(widgets=["model", "model.resolution"]),
+                expr="""
+                (
+                  $m := $lookup(widgets, "model");
+                  $res := $lookup(widgets, "model.resolution");
+                  $standard_model_prices := {
+                    "original": {"min": 0.01, "max": 0.198},
+                    "720p": {"min": 0.01, "max": 0.06},
+                    "1080p": {"min": 0.015, "max": 0.09},
+                    "2k/qhd": {"min": 0.02, "max": 0.117},
+                    "4k/uhd": {"min": 0.025, "max": 0.152},
+                    "8k": {"min": 0.033, "max": 0.198}
+                  };
+                  $ultra_hd_model_prices := {
+                    "original": {"min": 0.015, "max": 0.264},
+                    "720p": {"min": 0.015, "max": 0.092},
+                    "1080p": {"min": 0.02, "max": 0.12},
+                    "2k/qhd": {"min": 0.026, "max": 0.156},
+                    "4k/uhd": {"min": 0.034, "max": 0.203},
+                    "8k": {"min": 0.044, "max": 0.264}
+                  };
+                  $generative_model_prices := {
+                    "original": {"min": 0.015, "max": 0.338},
+                    "720p": {"min": 0.008, "max": 0.090},
+                    "1080p": {"min": 0.05, "max": 0.15},
+                    "2k/qhd": {"min": 0.038, "max": 0.225},
+                    "4k/uhd": {"min": 0.056, "max": 0.338}
+                  };
+                  $prices := $contains($m, "ultra hd") ? $ultra_hd_model_prices :
+                             $contains($m, "generative") ? $generative_model_prices :
+                             $standard_model_prices;
+                  $price := $lookup($prices, $res);
+                  {
+                    "type": "range_usd",
+                    "min_usd": $price.min,
+                    "max_usd": $price.max,
+                    "format": {"approximate": true, "suffix": "/second"}
+                  }
+                )
+                """,
+            ),
+        )
+
+    @classmethod
+    async def execute(
+        cls,
+        model: InputVideoModel,
+        video: Input.Video,
+    ) -> IO.NodeOutput:
+        validate_video_duration(video, min_duration=0.5, max_duration=60 * 60)
+        resolution = model["resolution"]
+        src_width, src_height = video.get_dimensions()
+
+        if resolution == "original":
+            output_width = src_width
+            output_height = src_height
+        else:
+            if src_width == src_height:
+                target_size = RESOLUTION_SQUARE_MAP[resolution]
+                if target_size < src_width:
+                    raise ValueError(
+                        f"Selected resolution {resolution} ({target_size}x{target_size}) is smaller than "
+                        f"the input video ({src_width}x{src_height}). Please select a higher resolution or 'original'."
+                    )
+                output_width = target_size
+                output_height = target_size
+            else:
+                min_dimension = min(src_width, src_height)
+                target_size = RESOLUTION_TARGET_MAP[resolution]
+                if target_size < min_dimension:
+                    raise ValueError(
+                        f"Selected resolution {resolution} ({target_size}p) is smaller than "
+                        f"the input video's shorter dimension ({min_dimension}p). "
+                        f"Please select a higher resolution or 'original'."
+                    )
+                if src_width > src_height:
+                    output_height = target_size
+                    output_width = int(target_size * (src_width / src_height))
+                else:
+                    output_width = target_size
+                    output_height = int(target_size * (src_height / src_width))
+        initial_res = await sync_op(
+            cls,
+            ApiEndpoint(path="/proxy/hitpaw/api/video-enhancer", method="POST"),
+            response_model=TaskCreateResponse,
+            data=VideoEnhanceTaskCreateRequest(
+                video_url=await upload_video_to_comfyapi(cls, video),
+                resolution=[output_width, output_height],
+                original_resolution=[src_width, src_height],
+                model_name=VIDEO_MODELS_MODELS_MAP[model["model"]],
+            ),
+            wait_label="Creating task",
+            final_label_on_success="Task created",
+        )
+        request_price = initial_res.data.consume_coins / 1000
+        if initial_res.code != 200:
+            raise ValueError(f"Task creation failed with code {initial_res.code}: {initial_res.message}")
+        final_response = await poll_op(
+            cls,
+            ApiEndpoint(path="/proxy/hitpaw/api/task-status", method="POST"),
+            data=TaskStatusPollRequest(job_id=initial_res.data.job_id),
+            response_model=TaskStatusResponse,
+            status_extractor=lambda x: x.data.status,
+            price_extractor=lambda x: request_price,
+            poll_interval=10.0,
+            max_poll_attempts=320,
+        )
+        return IO.NodeOutput(await download_url_to_video_output(final_response.data.res_url))
+
+
+class HitPawExtension(ComfyExtension):
+    @override
+    async def get_node_list(self) -> list[type[IO.ComfyNode]]:
+        return [
+            HitPawGeneralImageEnhance,
+            HitPawVideoEnhance,
+        ]
+
+
+async def comfy_entrypoint() -> HitPawExtension:
+    return HitPawExtension()
--- a/comfy_api_nodes/util/upload_helpers.py
+++ b/comfy_api_nodes/util/upload_helpers.py
@ -94,7 +94,7 @@ async def upload_image_to_comfyapi(
    *,
    mime_type: str | None = None,
    wait_label: str | None = "Uploading",
-    total_pixels: int = 2048 * 2048,
+    total_pixels: int | None = 2048 * 2048,
 ) -> str:
    """Uploads a single image to ComfyUI API and returns its download URL."""
    return (
--- a/comfy_execution/caching.py
+++ b/comfy_execution/caching.py
@ -23,7 +23,7 @@ def include_unique_id_in_input(class_type: str) -> bool:
    return NODE_CLASS_CONTAINS_UNIQUE_ID[class_type]

 class CacheKeySet(ABC):
-    def __init__(self, dynprompt, node_ids, is_changed_cache):
+    def __init__(self, dynprompt, node_ids, is_changed):
        self.keys = {}
        self.subcache_keys = {}

@ -45,6 +45,12 @@ class CacheKeySet(ABC):

    def get_subcache_key(self, node_id):
        return self.subcache_keys.get(node_id, None)
+    
+    async def update_cache_key(self, node_id) -> None:
+        pass
+
+    def is_key_updated(self, node_id) -> bool:
+        return True

 class Unhashable:
    def __init__(self):
@ -62,10 +68,21 @@ def to_hashable(obj):
    else:
        # TODO - Support other objects like tensors?
        return Unhashable()
+    
+def throw_on_unhashable(obj):
+    # Same as to_hashable except throwing for unhashables instead.
+    if isinstance(obj, (int, float, str, bool, bytes, type(None))):
+        return obj
+    elif isinstance(obj, Mapping):
+        return frozenset([(throw_on_unhashable(k), throw_on_unhashable(v)) for k, v in sorted(obj.items())])
+    elif isinstance(obj, Sequence):
+        return frozenset(zip(itertools.count(), [throw_on_unhashable(i) for i in obj]))
+    else:
+        raise Exception("Object unhashable.")

 class CacheKeySetID(CacheKeySet):
-    def __init__(self, dynprompt, node_ids, is_changed_cache):
-        super().__init__(dynprompt, node_ids, is_changed_cache)
+    def __init__(self, dynprompt, node_ids, is_changed):
+        super().__init__(dynprompt, node_ids, is_changed)
        self.dynprompt = dynprompt

    async def add_keys(self, node_ids):
@ -79,13 +96,25 @@ class CacheKeySetID(CacheKeySet):
            self.subcache_keys[node_id] = (node_id, node["class_type"])

 class CacheKeySetInputSignature(CacheKeySet):
-    def __init__(self, dynprompt, node_ids, is_changed_cache):
-        super().__init__(dynprompt, node_ids, is_changed_cache)
+    def __init__(self, dynprompt, node_ids, is_changed):
+        super().__init__(dynprompt, node_ids, is_changed)
        self.dynprompt = dynprompt
-        self.is_changed_cache = is_changed_cache
+        self.is_changed = is_changed
+        self.updated_node_ids = set()

    def include_node_id_in_input(self) -> bool:
        return False
+    
+    async def update_cache_key(self, node_id):
+        if node_id in self.updated_node_ids:
+            return
+        if node_id not in self.keys:
+            return
+        self.updated_node_ids.add(node_id)
+        self.keys[node_id] = await self.get_node_signature(node_id)
+
+    def is_key_updated(self, node_id):
+        return node_id in self.updated_node_ids

    async def add_keys(self, node_ids):
        for node_id in node_ids:
@ -94,28 +123,30 @@ class CacheKeySetInputSignature(CacheKeySet):
            if not self.dynprompt.has_node(node_id):
                continue
            node = self.dynprompt.get_node(node_id)
-            self.keys[node_id] = await self.get_node_signature(self.dynprompt, node_id)
+            self.keys[node_id] = None
            self.subcache_keys[node_id] = (node_id, node["class_type"])

-    async def get_node_signature(self, dynprompt, node_id):
-        signature = []
-        ancestors, order_mapping = self.get_ordered_ancestry(dynprompt, node_id)
-        signature.append(await self.get_immediate_node_signature(dynprompt, node_id, order_mapping))
+    async def get_node_signature(self, node_id):
+        signatures = []
+        ancestors, order_mapping, node_inputs = self.get_ordered_ancestry(node_id)
+        node = self.dynprompt.get_node(node_id)
+        node["signature"] = to_hashable(await self.get_immediate_node_signature(node_id, order_mapping, node_inputs))
+        signatures.append(node["signature"])
        for ancestor_id in ancestors:
-            signature.append(await self.get_immediate_node_signature(dynprompt, ancestor_id, order_mapping))
-        return to_hashable(signature)
+            ancestor_node = self.dynprompt.get_node(ancestor_id)
+            assert "signature" in ancestor_node
+            signatures.append(ancestor_node["signature"])
+        signatures = frozenset(zip(itertools.count(), signatures))
+        return signatures

-    async def get_immediate_node_signature(self, dynprompt, node_id, ancestor_order_mapping):
-        if not dynprompt.has_node(node_id):
+    async def get_immediate_node_signature(self, node_id, ancestor_order_mapping, inputs):
+        if not self.dynprompt.has_node(node_id):
            # This node doesn't exist -- we can't cache it.
            return [float("NaN")]
-        node = dynprompt.get_node(node_id)
+        node = self.dynprompt.get_node(node_id)
        class_type = node["class_type"]
        class_def = nodes.NODE_CLASS_MAPPINGS[class_type]
-        signature = [class_type, await self.is_changed_cache.get(node_id)]
-        if self.include_node_id_in_input() or (hasattr(class_def, "NOT_IDEMPOTENT") and class_def.NOT_IDEMPOTENT) or include_unique_id_in_input(class_type):
-            signature.append(node_id)
-        inputs = node["inputs"]
+        signature = [class_type, await self.is_changed.get(node_id)]
        for key in sorted(inputs.keys()):
            if is_link(inputs[key]):
                (ancestor_id, ancestor_socket) = inputs[key]
@ -123,28 +154,69 @@ class CacheKeySetInputSignature(CacheKeySet):
                signature.append((key,("ANCESTOR", ancestor_index, ancestor_socket)))
            else:
                signature.append((key, inputs[key]))
+        if self.include_node_id_in_input() or (hasattr(class_def, "NOT_IDEMPOTENT") and class_def.NOT_IDEMPOTENT) or include_unique_id_in_input(class_type):
+            signature.append(node_id)
        return signature
+    
+    def get_ordered_ancestry(self, node_id):
+        def get_ancestors(ancestors, ret: list=[]):
+            for ancestor_id in ancestors:
+                if ancestor_id not in ret:
+                    ret.append(ancestor_id)
+                ancestor_node = self.dynprompt.get_node(ancestor_id)
+                get_ancestors(ancestor_node["ancestors"], ret)
+            return ret
+        
+        ancestors, node_inputs = self.get_ordered_ancestry_internal(node_id)
+        ancestors = get_ancestors(ancestors)

-    # This function returns a list of all ancestors of the given node. The order of the list is
-    # deterministic based on which specific inputs the ancestor is connected by.
-    def get_ordered_ancestry(self, dynprompt, node_id):
-        ancestors = []
        order_mapping = {}
-        self.get_ordered_ancestry_internal(dynprompt, node_id, ancestors, order_mapping)
-        return ancestors, order_mapping
+        for i, ancestor_id in enumerate(ancestors):
+            order_mapping[ancestor_id] = i
+        
+        return ancestors, order_mapping, node_inputs

-    def get_ordered_ancestry_internal(self, dynprompt, node_id, ancestors, order_mapping):
-        if not dynprompt.has_node(node_id):
-            return
-        inputs = dynprompt.get_node(node_id)["inputs"]
-        input_keys = sorted(inputs.keys())
-        for key in input_keys:
-            if is_link(inputs[key]):
-                ancestor_id = inputs[key][0]
-                if ancestor_id not in order_mapping:
-                    ancestors.append(ancestor_id)
-                    order_mapping[ancestor_id] = len(ancestors) - 1
-                    self.get_ordered_ancestry_internal(dynprompt, ancestor_id, ancestors, order_mapping)
+    def get_ordered_ancestry_internal(self, node_id):
+        def get_hashable(obj):
+            try:
+                return throw_on_unhashable(obj)
+            except:
+                return Unhashable
+        
+        ancestors = []
+        node_inputs = {}
+
+        if not self.dynprompt.has_node(node_id):
+            return ancestors, node_inputs
+        
+        node = self.dynprompt.get_node(node_id)
+        if "ancestors" in node:
+            return node["ancestors"], node_inputs
+        
+        input_data_all, _, _ = self.is_changed.get_input_data(node_id)
+        inputs = self.dynprompt.get_node(node_id)["inputs"]
+        for key in sorted(inputs.keys()):
+            if key in input_data_all:
+                if is_link(inputs[key]):
+                    ancestor_id = inputs[key][0]
+                    hashable = get_hashable(input_data_all[key])
+                    if hashable is Unhashable or is_link(input_data_all[key][0]):
+                        # Link still needed
+                        node_inputs[key] = inputs[key]
+                        if ancestor_id not in ancestors:
+                            ancestors.append(ancestor_id)
+                    else:
+                        # Replace link
+                        node_inputs[key] = input_data_all[key]
+                else:
+                    hashable = get_hashable(inputs[key])
+                    if hashable is Unhashable:
+                        node_inputs[key] = Unhashable()
+                    else:
+                        node_inputs[key] = [inputs[key]]
+        
+        node["ancestors"] = ancestors
+        return node["ancestors"], node_inputs

 class BasicCache:
    def __init__(self, key_class):
@ -155,11 +227,11 @@ class BasicCache:
        self.cache = {}
        self.subcaches = {}

-    async def set_prompt(self, dynprompt, node_ids, is_changed_cache):
+    async def set_prompt(self, dynprompt, node_ids, is_changed):
        self.dynprompt = dynprompt
-        self.cache_key_set = self.key_class(dynprompt, node_ids, is_changed_cache)
+        self.cache_key_set = self.key_class(dynprompt, node_ids, is_changed)
        await self.cache_key_set.add_keys(node_ids)
-        self.is_changed_cache = is_changed_cache
+        self.is_changed = is_changed
        self.initialized = True

    def all_node_ids(self):
@ -185,6 +257,8 @@ class BasicCache:
        for key in self.subcaches:
            if key not in preserve_subcaches:
                to_remove.append(key)
+            else:
+                self.subcaches[key].clean_unused()
        for key in to_remove:
            del self.subcaches[key]

@ -196,16 +270,23 @@ class BasicCache:
    def poll(self, **kwargs):
        pass

+    async def _update_cache_key_immediate(self, node_id):
+        await self.cache_key_set.update_cache_key(node_id)
+    
+    def _is_key_updated_immediate(self, node_id):
+        return self.cache_key_set.is_key_updated(node_id)
+
    def _set_immediate(self, node_id, value):
        assert self.initialized
        cache_key = self.cache_key_set.get_data_key(node_id)
-        self.cache[cache_key] = value
+        if cache_key is not None:
+            self.cache[cache_key] = value

    def _get_immediate(self, node_id):
        if not self.initialized:
            return None
        cache_key = self.cache_key_set.get_data_key(node_id)
-        if cache_key in self.cache:
+        if cache_key is not None and cache_key in self.cache:
            return self.cache[cache_key]
        else:
            return None
@ -216,7 +297,7 @@ class BasicCache:
        if subcache is None:
            subcache = BasicCache(self.key_class)
            self.subcaches[subcache_key] = subcache
-        await subcache.set_prompt(self.dynprompt, children_ids, self.is_changed_cache)
+        await subcache.set_prompt(self.dynprompt, children_ids, self.is_changed)
        return subcache

    def _get_subcache(self, node_id):
@ -272,10 +353,20 @@ class HierarchicalCache(BasicCache):
        cache = self._get_cache_for(node_id)
        assert cache is not None
        return await cache._ensure_subcache(node_id, children_ids)
+    
+    async def update_cache_key(self, node_id):
+        cache = self._get_cache_for(node_id)
+        assert cache is not None
+        await cache._update_cache_key_immediate(node_id)
+        
+    def is_key_updated(self, node_id):
+        cache = self._get_cache_for(node_id)
+        assert cache is not None
+        return cache._is_key_updated_immediate(node_id)

 class NullCache:

-    async def set_prompt(self, dynprompt, node_ids, is_changed_cache):
+    async def set_prompt(self, dynprompt, node_ids, is_changed):
        pass

    def all_node_ids(self):
@ -295,6 +386,12 @@ class NullCache:

    async def ensure_subcache_for(self, node_id, children_ids):
        return self
+    
+    async def update_cache_key(self, node_id):
+        pass
+        
+    def is_key_updated(self, node_id):
+        return True

 class LRUCache(BasicCache):
    def __init__(self, key_class, max_size=100):
@ -305,8 +402,8 @@ class LRUCache(BasicCache):
        self.used_generation = {}
        self.children = {}

-    async def set_prompt(self, dynprompt, node_ids, is_changed_cache):
-        await super().set_prompt(dynprompt, node_ids, is_changed_cache)
+    async def set_prompt(self, dynprompt, node_ids, is_changed):
+        await super().set_prompt(dynprompt, node_ids, is_changed)
        self.generation += 1
        for node_id in node_ids:
            self._mark_used(node_id)
@ -314,7 +411,7 @@ class LRUCache(BasicCache):
    def clean_unused(self):
        while len(self.cache) > self.max_size and self.min_generation < self.generation:
            self.min_generation += 1
-            to_remove = [key for key in self.cache if self.used_generation[key] < self.min_generation]
+            to_remove = [key for key in self.cache if key not in self.used_generation or self.used_generation[key] < self.min_generation]
            for key in to_remove:
                del self.cache[key]
                del self.used_generation[key]
@ -347,6 +444,14 @@ class LRUCache(BasicCache):
            self._mark_used(child_id)
            self.children[cache_key].append(self.cache_key_set.get_data_key(child_id))
        return self
+    
+    async def update_cache_key(self, node_id):
+        await self._update_cache_key_immediate(node_id)
+        self._mark_used(node_id)
+        
+    def is_key_updated(self, node_id):
+        self._mark_used(node_id)
+        return self._is_key_updated_immediate(node_id)


 #Iterating the cache for usage analysis might be expensive, so if we trigger make sure
--- a/comfy_extras/nodes_ace.py
+++ b/comfy_extras/nodes_ace.py
@ -28,12 +28,39 @@ class TextEncodeAceStepAudio(io.ComfyNode):
        conditioning = node_helpers.conditioning_set_values(conditioning, {"lyrics_strength": lyrics_strength})
        return io.NodeOutput(conditioning)

+class TextEncodeAceStepAudio15(io.ComfyNode):
+    @classmethod
+    def define_schema(cls):
+        return io.Schema(
+            node_id="TextEncodeAceStepAudio1.5",
+            category="conditioning",
+            inputs=[
+                io.Clip.Input("clip"),
+                io.String.Input("tags", multiline=True, dynamic_prompts=True),
+                io.String.Input("lyrics", multiline=True, dynamic_prompts=True),
+                io.Int.Input("seed", default=0, min=0, max=0xffffffffffffffff, control_after_generate=True),
+                io.Int.Input("bpm", default=120, min=10, max=300),
+                io.Float.Input("duration", default=120.0, min=0.0, max=2000.0, step=0.1),
+                io.Combo.Input("timesignature", options=['2', '3', '4', '6']),
+                io.Combo.Input("language", options=["en", "ja", "zh", "es", "de", "fr", "pt", "ru", "it", "nl", "pl", "tr", "vi", "cs", "fa", "id", "ko", "uk", "hu", "ar", "sv", "ro", "el"]),
+                io.Combo.Input("keyscale", options=[f"{root} {quality}" for quality in ["major", "minor"] for root in ["C", "C#", "Db", "D", "D#", "Eb", "E", "F", "F#", "Gb", "G", "G#", "Ab", "A", "A#", "Bb", "B"]]),
+            ],
+            outputs=[io.Conditioning.Output()],
+        )
+
+    @classmethod
+    def execute(cls, clip, tags, lyrics, seed, bpm, duration, timesignature, language, keyscale) -> io.NodeOutput:
+        tokens = clip.tokenize(tags, lyrics=lyrics, bpm=bpm, duration=duration, timesignature=int(timesignature), language=language, keyscale=keyscale, seed=seed)
+        conditioning = clip.encode_from_tokens_scheduled(tokens)
+        return io.NodeOutput(conditioning)
+

 class EmptyAceStepLatentAudio(io.ComfyNode):
    @classmethod
    def define_schema(cls):
        return io.Schema(
            node_id="EmptyAceStepLatentAudio",
+            display_name="Empty Ace Step 1.0 Latent Audio",
            category="latent/audio",
            inputs=[
                io.Float.Input("seconds", default=120.0, min=1.0, max=1000.0, step=0.1),
@ -51,12 +78,60 @@ class EmptyAceStepLatentAudio(io.ComfyNode):
        return io.NodeOutput({"samples": latent, "type": "audio"})


+class EmptyAceStep15LatentAudio(io.ComfyNode):
+    @classmethod
+    def define_schema(cls):
+        return io.Schema(
+            node_id="EmptyAceStep1.5LatentAudio",
+            display_name="Empty Ace Step 1.5 Latent Audio",
+            category="latent/audio",
+            inputs=[
+                io.Float.Input("seconds", default=120.0, min=1.0, max=1000.0, step=0.01),
+                io.Int.Input(
+                    "batch_size", default=1, min=1, max=4096, tooltip="The number of latent images in the batch."
+                ),
+            ],
+            outputs=[io.Latent.Output()],
+        )
+
+    @classmethod
+    def execute(cls, seconds, batch_size) -> io.NodeOutput:
+        length = round((seconds * 48000 / 1920))
+        latent = torch.zeros([batch_size, 64, length], device=comfy.model_management.intermediate_device())
+        return io.NodeOutput({"samples": latent, "type": "audio"})
+
+class ReferenceTimbreAudio(io.ComfyNode):
+    @classmethod
+    def define_schema(cls):
+        return io.Schema(
+            node_id="ReferenceTimbreAudio",
+            category="advanced/conditioning/audio",
+            is_experimental=True,
+            description="This node sets the reference audio for timbre (for ace step 1.5)",
+            inputs=[
+                io.Conditioning.Input("conditioning"),
+                io.Latent.Input("latent", optional=True),
+            ],
+            outputs=[
+                io.Conditioning.Output(),
+            ]
+        )
+
+    @classmethod
+    def execute(cls, conditioning, latent=None) -> io.NodeOutput:
+        if latent is not None:
+            conditioning = node_helpers.conditioning_set_values(conditioning, {"reference_audio_timbre_latents": [latent["samples"]]}, append=True)
+        return io.NodeOutput(conditioning)
+
 class AceExtension(ComfyExtension):
    @override
    async def get_node_list(self) -> list[type[io.ComfyNode]]:
        return [
            TextEncodeAceStepAudio,
            EmptyAceStepLatentAudio,
+            TextEncodeAceStepAudio15,
+            EmptyAceStep15LatentAudio,
+            ReferenceTimbreAudio,
        ]

 async def comfy_entrypoint() -> AceExtension:
--- a/comfy_extras/nodes_audio.py
+++ b/comfy_extras/nodes_audio.py
@ -82,13 +82,14 @@ class VAEEncodeAudio(IO.ComfyNode):
    @classmethod
    def execute(cls, vae, audio) -> IO.NodeOutput:
        sample_rate = audio["sample_rate"]
-        if 44100 != sample_rate:
-            waveform = torchaudio.functional.resample(audio["waveform"], sample_rate, 44100)
+        vae_sample_rate = getattr(vae, "audio_sample_rate", 44100)
+        if vae_sample_rate != sample_rate:
+            waveform = torchaudio.functional.resample(audio["waveform"], sample_rate, vae_sample_rate)
        else:
            waveform = audio["waveform"]

        t = vae.encode(waveform.movedim(1, -1))
-        return IO.NodeOutput({"samples":t})
+        return IO.NodeOutput({"samples": t})

    encode = execute  # TODO: remove

@ -114,7 +115,8 @@ class VAEDecodeAudio(IO.ComfyNode):
        std = torch.std(audio, dim=[1,2], keepdim=True) * 5.0
        std[std < 1.0] = 1.0
        audio /= std
-        return IO.NodeOutput({"waveform": audio, "sample_rate": 44100 if "sample_rate" not in samples else samples["sample_rate"]})
+        vae_sample_rate = getattr(vae, "audio_sample_rate", 44100)
+        return IO.NodeOutput({"waveform": audio, "sample_rate": vae_sample_rate if "sample_rate" not in samples else samples["sample_rate"]})

    decode = execute  # TODO: remove

--- a/comfyui_version.py
+++ b/comfyui_version.py
@ -1,3 +1,3 @@
 # This file is automatically generated by the build process when version is
 # updated in pyproject.toml.
-__version__ = "0.11.1"
+__version__ = "0.12.0"
--- a/execution.py
+++ b/execution.py
@ -48,49 +48,40 @@ class ExecutionResult(Enum):
 class DuplicateNodeError(Exception):
    pass

-class IsChangedCache:
-    def __init__(self, prompt_id: str, dynprompt: DynamicPrompt, outputs_cache: BasicCache):
+class IsChanged:
+    def __init__(self, prompt_id: str, dynprompt: DynamicPrompt, execution_list: ExecutionList|None=None, extra_data: dict={}):
        self.prompt_id = prompt_id
        self.dynprompt = dynprompt
-        self.outputs_cache = outputs_cache
-        self.is_changed = {}
-
-    async def get(self, node_id):
-        if node_id in self.is_changed:
-            return self.is_changed[node_id]
+        self.execution_list = execution_list
+        self.extra_data = extra_data

+    def get_input_data(self, node_id):
+        node = self.dynprompt.get_node(node_id)
+        class_type = node["class_type"]
+        class_def = nodes.NODE_CLASS_MAPPINGS[class_type]
+        return get_input_data(node["inputs"], class_def, node_id, self.execution_list, self.dynprompt, self.extra_data)
+
+    async def get(self, node_id):
        node = self.dynprompt.get_node(node_id)
        class_type = node["class_type"]
        class_def = nodes.NODE_CLASS_MAPPINGS[class_type]
-        has_is_changed = False
        is_changed_name = None
        if issubclass(class_def, _ComfyNodeInternal) and first_real_override(class_def, "fingerprint_inputs") is not None:
-            has_is_changed = True
            is_changed_name = "fingerprint_inputs"
        elif hasattr(class_def, "IS_CHANGED"):
-            has_is_changed = True
            is_changed_name = "IS_CHANGED"
-        if not has_is_changed:
-            self.is_changed[node_id] = False
-            return self.is_changed[node_id]
+        if is_changed_name is None:
+            return False

-        if "is_changed" in node:
-            self.is_changed[node_id] = node["is_changed"]
-            return self.is_changed[node_id]
-
-        # Intentionally do not use cached outputs here. We only want constants in IS_CHANGED
-        input_data_all, _, v3_data = get_input_data(node["inputs"], class_def, node_id, None)
+        input_data_all, _, v3_data = self.get_input_data(node_id)
        try:
            is_changed = await _async_map_node_over_list(self.prompt_id, node_id, class_def, input_data_all, is_changed_name, v3_data=v3_data)
            is_changed = await resolve_map_node_over_list_results(is_changed)
-            node["is_changed"] = [None if isinstance(x, ExecutionBlocker) else x for x in is_changed]
+            is_changed = [None if isinstance(x, ExecutionBlocker) else x for x in is_changed]
        except Exception as e:
            logging.warning("WARNING: {}".format(e))
-            node["is_changed"] = float("NaN")
-        finally:
-            self.is_changed[node_id] = node["is_changed"]
-        return self.is_changed[node_id]
-
+            is_changed = float("NaN")
+        return is_changed

 class CacheEntry(NamedTuple):
    ui: dict
@ -416,16 +407,19 @@ async def execute(server, dynprompt, caches, current_item, extra_data, executed,
    inputs = dynprompt.get_node(unique_id)['inputs']
    class_type = dynprompt.get_node(unique_id)['class_type']
    class_def = nodes.NODE_CLASS_MAPPINGS[class_type]
-    cached = caches.outputs.get(unique_id)
-    if cached is not None:
-        if server.client_id is not None:
-            cached_ui = cached.ui or {}
-            server.send_sync("executed", { "node": unique_id, "display_node": display_node_id, "output": cached_ui.get("output",None), "prompt_id": prompt_id }, server.client_id)
-            if cached.ui is not None:
-                ui_outputs[unique_id] = cached.ui
-        get_progress_state().finish_progress(unique_id)
-        execution_list.cache_update(unique_id, cached)
-        return (ExecutionResult.SUCCESS, None, None)
+    if caches.outputs.is_key_updated(unique_id):
+        # Key is updated, the cache can be checked.
+        cached = caches.outputs.get(unique_id)
+        if cached is not None:
+            if server.client_id is not None:
+                cached_ui = cached.ui or {}
+                server.send_sync("execution_cached", { "nodes": [unique_id], "prompt_id": prompt_id}, server.client_id)
+                server.send_sync("executed", { "node": unique_id, "display_node": display_node_id, "output": cached_ui.get("output",None), "prompt_id": prompt_id }, server.client_id)
+                if cached.ui is not None:
+                    ui_outputs[unique_id] = cached.ui
+            get_progress_state().finish_progress(unique_id)
+            execution_list.cache_update(unique_id, cached)
+            return (ExecutionResult.SUCCESS, None, None)

    input_data_all = None
    try:
@ -466,11 +460,14 @@ async def execute(server, dynprompt, caches, current_item, extra_data, executed,
            del pending_subgraph_results[unique_id]
            has_subgraph = False
        else:
-            get_progress_state().start_progress(unique_id)
+            if caches.outputs.is_key_updated(unique_id):
+                # The key is updated, the node is executing.
+                get_progress_state().start_progress(unique_id)
+                if server.client_id is not None:
+                    server.last_node_id = display_node_id
+                    server.send_sync("executing", { "node": unique_id, "display_node": display_node_id, "prompt_id": prompt_id }, server.client_id)
+
            input_data_all, missing_keys, v3_data = get_input_data(inputs, class_def, unique_id, execution_list, dynprompt, extra_data)
-            if server.client_id is not None:
-                server.last_node_id = display_node_id
-                server.send_sync("executing", { "node": unique_id, "display_node": display_node_id, "prompt_id": prompt_id }, server.client_id)

            obj = caches.objects.get(unique_id)
            if obj is None:
@ -496,6 +493,14 @@ async def execute(server, dynprompt, caches, current_item, extra_data, executed,
                        execution_list.make_input_strong_link(unique_id, i)
                    return (ExecutionResult.PENDING, None, None)

+            if not caches.outputs.is_key_updated(unique_id):
+                # Update the cache key after any lazy inputs are evaluated.
+                async def update_cache_key(node_id, unblock):
+                    await caches.outputs.update_cache_key(node_id)
+                    unblock()
+                asyncio.create_task(update_cache_key(unique_id, execution_list.add_external_block(unique_id)))
+                return (ExecutionResult.PENDING, None, None)
+
            def execution_block_cb(block):
                if block.message is not None:
                    mes = {
@ -577,8 +582,7 @@ async def execute(server, dynprompt, caches, current_item, extra_data, executed,
                    cached_outputs.append((True, node_outputs))
            new_node_ids = set(new_node_ids)
            for cache in caches.all:
-                subcache = await cache.ensure_subcache_for(unique_id, new_node_ids)
-                subcache.clean_unused()
+                await cache.ensure_subcache_for(unique_id, new_node_ids)
            for node_id in new_output_ids:
                execution_list.add_node(node_id)
                execution_list.cache_link(node_id, unique_id)
@ -703,25 +707,16 @@ class PromptExecutor:
            dynamic_prompt = DynamicPrompt(prompt)
            reset_progress_state(prompt_id, dynamic_prompt)
            add_progress_handler(WebUIProgressHandler(self.server))
-            is_changed_cache = IsChangedCache(prompt_id, dynamic_prompt, self.caches.outputs)
+            execution_list = ExecutionList(dynamic_prompt, self.caches.outputs)
+            is_changed = IsChanged(prompt_id, dynamic_prompt, execution_list, extra_data)
            for cache in self.caches.all:
-                await cache.set_prompt(dynamic_prompt, prompt.keys(), is_changed_cache)
-                cache.clean_unused()
-
-            cached_nodes = []
-            for node_id in prompt:
-                if self.caches.outputs.get(node_id) is not None:
-                    cached_nodes.append(node_id)
+                await cache.set_prompt(dynamic_prompt, prompt.keys(), is_changed)

            comfy.model_management.cleanup_models_gc()
-            self.add_message("execution_cached",
-                          { "nodes": cached_nodes, "prompt_id": prompt_id},
-                          broadcast=False)
            pending_subgraph_results = {}
            pending_async_nodes = {} # TODO - Unify this with pending_subgraph_results
            ui_node_outputs = {}
            executed = set()
-            execution_list = ExecutionList(dynamic_prompt, self.caches.outputs)
            current_outputs = self.caches.outputs.all_node_ids()
            for node_id in list(execute_outputs):
                execution_list.add_node(node_id)
@ -759,7 +754,9 @@ class PromptExecutor:
            self.server.last_node_id = None
            if comfy.model_management.DISABLE_SMART_MEMORY:
                comfy.model_management.unload_all_models()
-
+            
+            for cache in self.caches.all:
+                cache.clean_unused()

 async def validate_inputs(prompt_id, prompt, item, validated):
    unique_id = item
--- a/nodes.py
+++ b/nodes.py
@ -1001,7 +1001,7 @@ class DualCLIPLoader:
    def INPUT_TYPES(s):
        return {"required": { "clip_name1": (folder_paths.get_filename_list("text_encoders"), ),
                              "clip_name2": (folder_paths.get_filename_list("text_encoders"), ),
-                              "type": (["sdxl", "sd3", "flux", "hunyuan_video", "hidream", "hunyuan_image", "hunyuan_video_15", "kandinsky5", "kandinsky5_image", "ltxv", "newbie"], ),
+                              "type": (["sdxl", "sd3", "flux", "hunyuan_video", "hidream", "hunyuan_image", "hunyuan_video_15", "kandinsky5", "kandinsky5_image", "ltxv", "newbie", "ace"], ),
                              },
                "optional": {
                              "device": (["default", "cpu"], {"advanced": True}),
--- a/pyproject.toml
+++ b/pyproject.toml
@ -1,6 +1,6 @@
 [project]
 name = "ComfyUI"
-version = "0.11.1"
+version = "0.12.0"
 readme = "README.md"
 license = { file = "LICENSE" }
 requires-python = ">=3.10"
--- a/requirements.txt
+++ b/requirements.txt
@ -1,5 +1,5 @@
 comfyui-frontend-package==1.37.11
-comfyui-workflow-templates==0.8.27
+comfyui-workflow-templates==0.8.31
 comfyui-embedded-docs==0.4.0
 torch
 torchsde
--- a/tests/execution/test_execution.py
+++ b/tests/execution/test_execution.py
@ -552,27 +552,50 @@ class TestExecution:
        assert len(images1) == 1, "Should have 1 image"
        assert len(images2) == 1, "Should have 1 image"

-    # This tests that only constant outputs are used in the call to `IS_CHANGED`
-    def test_is_changed_with_outputs(self, client: ComfyClient, builder: GraphBuilder, server):
+    def test_is_changed_passed_cached_outputs(self, client: ComfyClient, builder: GraphBuilder, server):
        g = builder
        input1 = g.node("StubConstantImage", value=0.5, height=512, width=512, batch_size=1)
-        test_node = g.node("TestIsChangedWithConstants", image=input1.out(0), value=0.5)
-
+        test_node = g.node("TestIsChangedWithAllInputs", image=input1.out(0), value=0.5)
        output = g.node("PreviewImage", images=test_node.out(0))

-        result = client.run(g)
-        images = result.get_images(output)
+        result1 = client.run(g)
+        images = result1.get_images(output)
        assert len(images) == 1, "Should have 1 image"
        assert numpy.array(images[0]).min() == 63 and numpy.array(images[0]).max() == 63, "Image should have value 0.25"

-        result = client.run(g)
-        images = result.get_images(output)
+        result2 = client.run(g)
+        images = result2.get_images(output)
        assert len(images) == 1, "Should have 1 image"
        assert numpy.array(images[0]).min() == 63 and numpy.array(images[0]).max() == 63, "Image should have value 0.25"
+        
        if server["should_cache_results"]:
-            assert not result.did_run(test_node), "The execution should have been cached"
+            assert not result2.did_run(test_node), "Test node should not have run again"
        else:
-            assert result.did_run(test_node), "The execution should have been re-run"
+            assert result2.did_run(test_node), "Test node should always run here"
+
+    def test_dont_always_run_downstream(self, client: ComfyClient, builder: GraphBuilder, server):
+        g = builder
+        float1 = g.node("TestDontAlwaysRunDownstream", float=0.5) # IS_CHANGED returns float("NaN")
+        image1 = g.node("StubConstantImage", value=float1.out(0), height=512, width=512, batch_size=1)
+        output = g.node("PreviewImage", images=image1.out(0))
+
+        result1 = client.run(g)
+        images = result1.get_images(output)
+        assert len(images) == 1, "Should have 1 image"
+        assert numpy.array(images[0]).min() == 127 and numpy.array(images[0]).max() == 127, "Image should have value 0.50"
+
+        result2 = client.run(g)
+        images = result2.get_images(output)
+        assert len(images) == 1, "Should have 1 image"
+        assert numpy.array(images[0]).min() == 127 and numpy.array(images[0]).max() == 127, "Image should have value 0.50"
+
+        assert result2.did_run(float1), "Float node should always run"
+        if server["should_cache_results"]:
+            assert not result2.did_run(image1), "Image node should not have run again"
+            assert not result2.did_run(output), "Output node should not have run again"
+        else:
+            assert result2.did_run(image1), "Image node should have run again"
+            assert result2.did_run(output), "Output node should have run again"


    def test_parallel_sleep_nodes(self, client: ComfyClient, builder: GraphBuilder, skip_timing_checks):
--- a/tests/execution/testing_nodes/testing-pack/specific_tests.py
+++ b/tests/execution/testing_nodes/testing-pack/specific_tests.py
@ -100,7 +100,7 @@ class TestCustomIsChanged:
        else:
            return False

-class TestIsChangedWithConstants:
+class TestIsChangedWithAllInputs:
    @classmethod
    def INPUT_TYPES(cls):
        return {
@ -120,10 +120,29 @@ class TestIsChangedWithConstants:

    @classmethod
    def IS_CHANGED(cls, image, value):
-        if image is None:
-            return value
-        else:
-            return image.mean().item() * value
+        # if image is None then an exception is thrown and is_changed becomes float("NaN")
+        return image.mean().item() * value
+    
+class TestDontAlwaysRunDownstream:
+    @classmethod
+    def INPUT_TYPES(cls):
+        return {
+            "required": {
+                "float": ("FLOAT",),
+            },
+        }
+    
+    RETURN_TYPES = ("FLOAT",)
+    FUNCTION = "always_run"
+
+    CATEGORY = "Testing/Nodes"
+
+    def always_run(self, float):
+        return (float,)
+    
+    @classmethod
+    def IS_CHANGED(cls, *args, **kwargs):
+        return float("NaN")

 class TestCustomValidation1:
    @classmethod
@ -486,7 +505,8 @@ TEST_NODE_CLASS_MAPPINGS = {
    "TestLazyMixImages": TestLazyMixImages,
    "TestVariadicAverage": TestVariadicAverage,
    "TestCustomIsChanged": TestCustomIsChanged,
-    "TestIsChangedWithConstants": TestIsChangedWithConstants,
+    "TestIsChangedWithAllInputs": TestIsChangedWithAllInputs,
+    "TestDontAlwaysRunDownstream": TestDontAlwaysRunDownstream,
    "TestCustomValidation1": TestCustomValidation1,
    "TestCustomValidation2": TestCustomValidation2,
    "TestCustomValidation3": TestCustomValidation3,
@ -504,7 +524,8 @@ TEST_NODE_DISPLAY_NAME_MAPPINGS = {
    "TestLazyMixImages": "Lazy Mix Images",
    "TestVariadicAverage": "Variadic Average",
    "TestCustomIsChanged": "Custom IsChanged",
-    "TestIsChangedWithConstants": "IsChanged With Constants",
+    "TestIsChangedWithAllInputs": "IsChanged With All Inputs",
+    "TestDontAlwaysRunDownstream": "Dont Always Run Downstream",
    "TestCustomValidation1": "Custom Validation 1",
    "TestCustomValidation2": "Custom Validation 2",
    "TestCustomValidation3": "Custom Validation 3",
Author	SHA1	Message	Date
Deluxe233	062183cba9	Merge `2db3ca609d` into `f5030e26fd`	2026-02-03 04:21:38 -05:00
comfyanonymous	f5030e26fd	Add progress bar to ace step. (#12242 ) Some checks failed Python Linting / Run Ruff (push) Waiting to run Details Python Linting / Run Pylint (push) Waiting to run Details Full Comfy CI Workflow Runs / test-stable (12.1, , linux, 3.10, [self-hosted Linux], stable) (push) Waiting to run Details Full Comfy CI Workflow Runs / test-stable (12.1, , linux, 3.11, [self-hosted Linux], stable) (push) Waiting to run Details Full Comfy CI Workflow Runs / test-stable (12.1, , linux, 3.12, [self-hosted Linux], stable) (push) Waiting to run Details Full Comfy CI Workflow Runs / test-unix-nightly (12.1, , linux, 3.11, [self-hosted Linux], nightly) (push) Waiting to run Details Execution Tests / test (macos-latest) (push) Waiting to run Details Execution Tests / test (ubuntu-latest) (push) Waiting to run Details Execution Tests / test (windows-latest) (push) Waiting to run Details Test server launches without errors / test (push) Waiting to run Details Unit Tests / test (macos-latest) (push) Waiting to run Details Unit Tests / test (ubuntu-latest) (push) Waiting to run Details Unit Tests / test (windows-2022) (push) Waiting to run Details Build package / Build Test (3.10) (push) Has been cancelled Details Build package / Build Test (3.11) (push) Has been cancelled Details Build package / Build Test (3.12) (push) Has been cancelled Details Build package / Build Test (3.13) (push) Has been cancelled Details Build package / Build Test (3.14) (push) Has been cancelled Details	2026-02-03 04:09:30 -05:00
Deluxe233	2db3ca609d	Simplified storing signatures/ancestors	2026-02-03 03:01:52 -05:00
comfyanonymous	66e1b07402	ComfyUI v0.12.0	2026-02-03 02:20:59 -05:00
Deluxe233	982092f79a	Removed unnecessary changes	2026-02-03 02:11:48 -05:00
ComfyUI Wiki	be4345d1c9	chore: update workflow templates to v0.8.31 (#12239 )	2026-02-02 23:08:43 -08:00
comfyanonymous	3c1a1a2df8	Basic support for the ace step 1.5 model. (#12237 )	2026-02-03 00:06:18 -05:00
Alexander Piskun	ba5bf3f1a8	[API Nodes] HitPaw API nodes (#12117 ) * feat(api-nodes): add HitPaw API nodes * remove face_soft_2x model as not working --------- Co-authored-by: Robin Huang <robin.j.huang@gmail.com>	2026-02-02 19:17:59 -08:00
comfyanonymous	c05a08ae66	Add back function. (#12234 ) Some checks are pending Python Linting / Run Ruff (push) Waiting to run Details Python Linting / Run Pylint (push) Waiting to run Details Full Comfy CI Workflow Runs / test-stable (12.1, , linux, 3.10, [self-hosted Linux], stable) (push) Waiting to run Details Full Comfy CI Workflow Runs / test-stable (12.1, , linux, 3.11, [self-hosted Linux], stable) (push) Waiting to run Details Full Comfy CI Workflow Runs / test-stable (12.1, , linux, 3.12, [self-hosted Linux], stable) (push) Waiting to run Details Full Comfy CI Workflow Runs / test-unix-nightly (12.1, , linux, 3.11, [self-hosted Linux], nightly) (push) Waiting to run Details Execution Tests / test (macos-latest) (push) Waiting to run Details Execution Tests / test (ubuntu-latest) (push) Waiting to run Details Execution Tests / test (windows-latest) (push) Waiting to run Details Test server launches without errors / test (push) Waiting to run Details Unit Tests / test (macos-latest) (push) Waiting to run Details Unit Tests / test (ubuntu-latest) (push) Waiting to run Details Unit Tests / test (windows-2022) (push) Waiting to run Details	2026-02-02 19:52:07 -05:00
rattus	de9ada6a41	Dynamic VRAM unloading fix (#12227 ) * mp: fix full dynamic unloading This was not unloading dynamic models when requesting a full unload via the unpatch() code path. This was ok, i your workflow was all dynamic models but fails with big VRAM leaks if you need to fully unload something for a regular ModelPatcher It also fices the "unload models" button. * mm: load models outside of Aimdo Mempool In dynamic_vram mode, escape the Aimdo mempool and load into the regular mempool. Use a dummy thread to do it.	2026-02-02 17:35:20 -05:00
rattus	37f711d4a1	mm: Fix cast buffers with intel offloading (#12229 ) Intel has offloading support but there were some nvidia calls in the new cast buffer stuff.	2026-02-02 17:34:46 -05:00
Deluxe233	90f57e6a8d	Fix not cleaning subcaches	2026-01-31 07:06:41 -05:00
Deluxe233	5bad474118	fix signature inconsistency	2026-01-30 22:25:02 -05:00
Deluxe233	c6b6128b2b	Fix issue with subcache's cache	2026-01-30 13:21:02 -05:00
Deluxe233	3770dc0ec4	tweak test	2026-01-29 05:39:23 -05:00
Deluxe233	96e9a81cdf	Fix not taking rawLink into account Forgot that input_data_all puts everything in a list.	2026-01-29 03:57:41 -05:00
Deluxe233	5cf4115f50	Added "execution_cached" message back in	2026-01-29 01:22:33 -05:00
Deluxe233	b951181123	Added tests + cleanup	2026-01-28 10:37:10 -05:00
Deluxe233	af4d691d1f	Revert "Included original cache key set for testing" This reverts commit `f511703343`.	2026-01-27 20:57:44 -05:00
Deluxe233	f511703343	Included original cache key set for testing	2026-01-27 13:51:41 -05:00
Deluxe233	1107f4322b	Removed unused method	2026-01-26 09:40:45 -05:00
Deluxe233	4683136740	Update caching.py	2026-01-26 09:33:00 -05:00
Deluxe233	38ab4e3c76	Fixed not taking rawLink into account.	2026-01-26 03:50:50 -05:00
Deluxe233	232995856e	Added a new type of cache key set.	2026-01-26 01:28:43 -05:00