Merge branch 'comfyanonymous:master' into master

kinqsradio · web-flow · commit b1e27d1c02b5 · 2024-08-30T13:37:10.000+10:00
diff --git a/.ci/windows_base_files/README_VERY_IMPORTANT.txt b/.ci/windows_base_files/README_VERY_IMPORTANT.txt
@@ -14,7 +14,7 @@ run_cpu.bat
 
 IF YOU GET A RED ERROR IN THE UI MAKE SURE YOU HAVE A MODEL/CHECKPOINT IN: ComfyUI\models\checkpoints
 
-You can download the stable diffusion 1.5 one from: https://huggingface.co/runwayml/stable-diffusion-v1-5/blob/main/v1-5-pruned-emaonly.ckpt
+You can download the stable diffusion 1.5 one from: https://huggingface.co/Comfy-Org/stable-diffusion-v1-5-archive/blob/main/v1-5-pruned-emaonly-fp16.safetensors
 
 
 RECOMMENDED WAY TO UPDATE:
diff --git a/comfy/controlnet.py b/comfy/controlnet.py
@@ -148,7 +148,7 @@ def control_merge(self, control, control_prev, output_dtype):
                         elif self.strength_type == StrengthType.LINEAR_UP:
                             x *= (self.strength ** float(len(control_output) - i))
 
-                    if x.dtype != output_dtype:
+                    if output_dtype is not None and x.dtype != output_dtype:
                         x = x.to(output_dtype)
 
                 out[key].append(x)
@@ -206,7 +206,6 @@ def get_control(self, x_noisy, t, cond, batched_number):
         if self.manual_cast_dtype is not None:
             dtype = self.manual_cast_dtype
 
-        output_dtype = x_noisy.dtype
         if self.cond_hint is None or x_noisy.shape[2] * self.compression_ratio != self.cond_hint.shape[2] or x_noisy.shape[3] * self.compression_ratio != self.cond_hint.shape[3]:
             if self.cond_hint is not None:
                 del self.cond_hint
@@ -236,7 +235,7 @@ def get_control(self, x_noisy, t, cond, batched_number):
         x_noisy = self.model_sampling_current.calculate_input(t, x_noisy)
 
         control = self.control_model(x=x_noisy.to(dtype), hint=self.cond_hint, timesteps=timestep.to(dtype), context=context.to(dtype), **extra)
-        return self.control_merge(control, control_prev, output_dtype)
+        return self.control_merge(control, control_prev, output_dtype=None)
 
     def copy(self):
         c = ControlNet(None, global_average_pooling=self.global_average_pooling, load_device=self.load_device, manual_cast_dtype=self.manual_cast_dtype)
@@ -445,7 +444,12 @@ def load_controlnet_flux_instantx(sd):
     for k in sd:
         new_sd[k] = sd[k]
 
-    control_model = comfy.ldm.flux.controlnet.ControlNetFlux(latent_input=True, operations=operations, device=offload_device, dtype=unet_dtype, **model_config.unet_config)
+    num_union_modes = 0
+    union_cnet = "controlnet_mode_embedder.weight"
+    if union_cnet in new_sd:
+        num_union_modes = new_sd[union_cnet].shape[0]
+
+    control_model = comfy.ldm.flux.controlnet.ControlNetFlux(latent_input=True, num_union_modes=num_union_modes, operations=operations, device=offload_device, dtype=unet_dtype, **model_config.unet_config)
     control_model = controlnet_load_state_dict(control_model, new_sd)
 
     latent_format = comfy.latent_formats.Flux()
diff --git a/comfy/ldm/flux/controlnet.py b/comfy/ldm/flux/controlnet.py
@@ -14,7 +14,7 @@
 
 
 class ControlNetFlux(Flux):
-    def __init__(self, latent_input=False, image_model=None, dtype=None, device=None, operations=None, **kwargs):
+    def __init__(self, latent_input=False, num_union_modes=0, image_model=None, dtype=None, device=None, operations=None, **kwargs):
         super().__init__(final_layer=False, dtype=dtype, device=device, operations=operations, **kwargs)
 
         self.main_model_double = 19
@@ -23,8 +23,17 @@ def __init__(self, latent_input=False, image_model=None, dtype=None, device=None
         self.controlnet_blocks = nn.ModuleList([])
         for _ in range(self.params.depth):
             controlnet_block = operations.Linear(self.hidden_size, self.hidden_size, dtype=dtype, device=device)
-            # controlnet_block = zero_module(controlnet_block)
             self.controlnet_blocks.append(controlnet_block)
+
+        self.controlnet_single_blocks = nn.ModuleList([])
+        for _ in range(self.params.depth_single_blocks):
+            self.controlnet_single_blocks.append(operations.Linear(self.hidden_size, self.hidden_size, dtype=dtype, device=device))
+
+        self.num_union_modes = num_union_modes
+        self.controlnet_mode_embedder = None
+        if self.num_union_modes > 0:
+            self.controlnet_mode_embedder = operations.Embedding(self.num_union_modes, self.hidden_size, dtype=dtype, device=device)
+
         self.gradient_checkpointing = False
         self.latent_input = latent_input
         self.pos_embed_input = operations.Linear(self.in_channels, self.hidden_size, bias=True, dtype=dtype, device=device)
@@ -57,6 +66,7 @@ def forward_orig(
         timesteps: Tensor,
         y: Tensor,
         guidance: Tensor = None,
+        control_type: Tensor = None,
     ) -> Tensor:
         if img.ndim != 3 or txt.ndim != 3:
             raise ValueError("Input img and txt tensors must have 3 dimensions.")
@@ -75,29 +85,47 @@ def forward_orig(
         vec = vec + self.vector_in(y)
         txt = self.txt_in(txt)
 
+        if self.controlnet_mode_embedder is not None and len(control_type) > 0:
+            control_cond = self.controlnet_mode_embedder(torch.tensor(control_type, device=img.device), out_dtype=img.dtype).unsqueeze(0).repeat((txt.shape[0], 1, 1))
+            txt = torch.cat([control_cond, txt], dim=1)
+            txt_ids = torch.cat([txt_ids[:,:1], txt_ids], dim=1)
+
         ids = torch.cat((txt_ids, img_ids), dim=1)
         pe = self.pe_embedder(ids)
 
-        block_res_samples = ()
+        controlnet_double = ()
+
+        for i in range(len(self.double_blocks)):
+            img, txt = self.double_blocks[i](img=img, txt=txt, vec=vec, pe=pe)
+            controlnet_double = controlnet_double + (self.controlnet_blocks[i](img),)
 
-        for block in self.double_blocks:
-            img, txt = block(img=img, txt=txt, vec=vec, pe=pe)
-            block_res_samples = block_res_samples + (img,)
+        img = torch.cat((txt, img), 1)
 
-        controlnet_block_res_samples = ()
-        for block_res_sample, controlnet_block in zip(block_res_samples, self.controlnet_blocks):
-            block_res_sample = controlnet_block(block_res_sample)
-            controlnet_block_res_samples = controlnet_block_res_samples + (block_res_sample,)
+        controlnet_single = ()
 
+        for i in range(len(self.single_blocks)):
+            img = self.single_blocks[i](img, vec=vec, pe=pe)
+            controlnet_single = controlnet_single + (self.controlnet_single_blocks[i](img[:, txt.shape[1] :, ...]),)
 
-        repeat = math.ceil(self.main_model_double / len(controlnet_block_res_samples))
+        repeat = math.ceil(self.main_model_double / len(controlnet_double))
         if self.latent_input:
             out_input = ()
-            for x in controlnet_block_res_samples:
+            for x in controlnet_double:
                     out_input += (x,) * repeat
         else:
-            out_input = (controlnet_block_res_samples * repeat)
-        return {"input": out_input[:self.main_model_double]}
+            out_input = (controlnet_double * repeat)
+
+        out = {"input": out_input[:self.main_model_double]}
+        if len(controlnet_single) > 0:
+            repeat = math.ceil(self.main_model_single / len(controlnet_single))
+            out_output = ()
+            if self.latent_input:
+                for x in controlnet_single:
+                        out_output += (x,) * repeat
+            else:
+                out_output = (controlnet_single * repeat)
+            out["output"] = out_output[:self.main_model_single]
+        return out
 
     def forward(self, x, timesteps, context, y, guidance=None, hint=None, **kwargs):
         patch_size = 2
@@ -120,4 +148,4 @@ def forward(self, x, timesteps, context, y, guidance=None, hint=None, **kwargs):
         img_ids = repeat(img_ids, "h w c -> b (h w) c", b=bs)
 
         txt_ids = torch.zeros((bs, context.shape[1], 3), device=x.device, dtype=x.dtype)
-        return self.forward_orig(img, img_ids, hint, context, txt_ids, timesteps, y, guidance)
+        return self.forward_orig(img, img_ids, hint, context, txt_ids, timesteps, y, guidance, control_type=kwargs.get("control_type", []))
diff --git a/comfy/lora.py b/comfy/lora.py
@@ -540,7 +540,7 @@ def calculate_weight(patches, weight, key, intermediate_dtype=torch.float32):
             b2 = comfy.model_management.cast_to_device(v[3].flatten(start_dim=1), weight.device, intermediate_dtype)
 
             try:
-                lora_diff = (torch.mm(b2, b1) + torch.mm(torch.mm(weight.flatten(start_dim=1), a2), a1)).reshape(weight.shape)
+                lora_diff = (torch.mm(b2, b1) + torch.mm(torch.mm(weight.flatten(start_dim=1).to(dtype=intermediate_dtype), a2), a1)).reshape(weight.shape)
                 if dora_scale is not None:
                     weight = function(weight_decompose(dora_scale, weight, lora_diff, alpha, strength, intermediate_dtype))
                 else:
diff --git a/notebooks/comfyui_colab.ipynb b/notebooks/comfyui_colab.ipynb
@@ -79,7 +79,7 @@
         "#!wget -c https://huggingface.co/comfyanonymous/clip_vision_g/resolve/main/clip_vision_g.safetensors -P ./models/clip_vision/\n",
         "\n",
         "# SD1.5\n",
-        "!wget -c https://huggingface.co/runwayml/stable-diffusion-v1-5/resolve/main/v1-5-pruned-emaonly.ckpt -P ./models/checkpoints/\n",
+        "!wget -c https://huggingface.co/Comfy-Org/stable-diffusion-v1-5-archive/resolve/main/v1-5-pruned-emaonly-fp16.safetensors -P ./models/checkpoints/\n",
         "\n",
         "# SD2\n",
         "#!wget -c https://huggingface.co/stabilityai/stable-diffusion-2-1-base/resolve/main/v2-1_512-ema-pruned.safetensors -P ./models/checkpoints/\n",
diff --git a/script_examples/basic_api_example.py b/script_examples/basic_api_example.py
@@ -43,7 +43,7 @@
     "4": {
         "class_type": "CheckpointLoaderSimple",
         "inputs": {
-            "ckpt_name": "v1-5-pruned-emaonly.ckpt"
+            "ckpt_name": "v1-5-pruned-emaonly.safetensors"
         }
     },
     "5": {
diff --git a/script_examples/websockets_api_example.py b/script_examples/websockets_api_example.py
@@ -84,7 +84,7 @@ def get_images(ws, prompt):
     "4": {
         "class_type": "CheckpointLoaderSimple",
         "inputs": {
-            "ckpt_name": "v1-5-pruned-emaonly.ckpt"
+            "ckpt_name": "v1-5-pruned-emaonly.safetensors"
         }
     },
     "5": {
diff --git a/script_examples/websockets_api_example_ws_images.py b/script_examples/websockets_api_example_ws_images.py
@@ -81,7 +81,7 @@ def get_images(ws, prompt):
     "4": {
         "class_type": "CheckpointLoaderSimple",
         "inputs": {
-            "ckpt_name": "v1-5-pruned-emaonly.ckpt"
+            "ckpt_name": "v1-5-pruned-emaonly.safetensors"
         }
     },
     "5": {

Original file line number	Diff line number	Diff line change
`@@ -43,7 +43,7 @@`
`43`	`43`	`"4": {`
`44`	`44`	`"class_type": "CheckpointLoaderSimple",`
`45`	`45`	`"inputs": {`
`46`		`- "ckpt_name": "v1-5-pruned-emaonly.ckpt"`
	`46`	`+ "ckpt_name": "v1-5-pruned-emaonly.safetensors"`
`47`	`47`	`}`
`48`	`48`	`},`
`49`	`49`	`"5": {`
Original file line number	Diff line number	Diff line change
`@@ -84,7 +84,7 @@ def get_images(ws, prompt):`
`84`	`84`	`"4": {`
`85`	`85`	`"class_type": "CheckpointLoaderSimple",`
`86`	`86`	`"inputs": {`
`87`		`- "ckpt_name": "v1-5-pruned-emaonly.ckpt"`
	`87`	`+ "ckpt_name": "v1-5-pruned-emaonly.safetensors"`
`88`	`88`	`}`
`89`	`89`	`},`
`90`	`90`	`"5": {`
Original file line number	Diff line number	Diff line change
`@@ -81,7 +81,7 @@ def get_images(ws, prompt):`
`81`	`81`	`"4": {`
`82`	`82`	`"class_type": "CheckpointLoaderSimple",`
`83`	`83`	`"inputs": {`
`84`		`- "ckpt_name": "v1-5-pruned-emaonly.ckpt"`
	`84`	`+ "ckpt_name": "v1-5-pruned-emaonly.safetensors"`
`85`	`85`	`}`
`86`	`86`	`},`
`87`	`87`	`"5": {`