From 5fbfe8c9fdaa0040301155e2993ce61a26d4a9af Mon Sep 17 00:00:00 2001 From: li-lizhe <147392333@qq.com> Date: Mon, 14 Sep 2026 09:15:19 +0800 Subject: [PATCH 1/2] fix(wan): use device-agnostic default for get_i2v_mask The `get_i2v_mask` method had a hardcoded `device="cuda"` default, which crashes on non-CUDA accelerators (Ascend NPU, etc.) with "Torch not compiled with CUDA enabled" when called without an explicit device argument. Change the default to None and resolve via `self._execution_device`, matching the pattern used across other pipeline methods. Verified on Ascend 910B NPU: torch.zeros(device="cuda") crashes, fix with device-agnostic resolution creates tensors on the correct device. --- src/diffusers/pipelines/wan/pipeline_wan_animate.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/diffusers/pipelines/wan/pipeline_wan_animate.py b/src/diffusers/pipelines/wan/pipeline_wan_animate.py index a923219a7550..c2416243f111 100644 --- a/src/diffusers/pipelines/wan/pipeline_wan_animate.py +++ b/src/diffusers/pipelines/wan/pipeline_wan_animate.py @@ -465,8 +465,10 @@ def get_i2v_mask( mask_len: int = 1, mask_pixel_values: torch.Tensor | None = None, dtype: torch.dtype | None = None, - device: str | torch.device = "cuda", + device: str | torch.device | None = None, ) -> torch.Tensor: + device = device or self._execution_device + # mask_pixel_values shape (if supplied): [B, C = 1, T, latent_h, latent_w] if mask_pixel_values is None: mask_lat_size = torch.zeros( From cf24c45cfea7dcfad3e4bea9df6964b8a66a323c Mon Sep 17 00:00:00 2001 From: li-lizhe <147392333@qq.com> Date: Mon, 14 Sep 2026 01:15:19 +0000 Subject: [PATCH 2/2] fix(wan_animate_2): drop hardcoded `device="cuda"` default in `get_i2v_mask` `get_i2v_mask` in `modular_pipelines/wan_animate_2/encoders.py` defaulted its `device` argument to `"cuda"`, so any call that omitted it allocated the mask on CUDA and failed on non-CUDA accelerators (NPU/XPU/MPS/CPU) with `AssertionError: Torch not compiled with CUDA enabled`. The default is now `None` and raises a clear `ValueError` instead of silently allocating on CUDA. All in-tree call sites already pass `device=` explicitly (`wan_animate_2/denoise.py` lines 128 and 220), so behaviour for existing callers is unchanged. Same fix as the one applied to `WanAnimatePipeline.get_i2v_mask` in this PR. --- .../modular_pipelines/wan_animate_2/encoders.py | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/src/diffusers/modular_pipelines/wan_animate_2/encoders.py b/src/diffusers/modular_pipelines/wan_animate_2/encoders.py index 21b70f636f7d..3eb85b41fe4c 100644 --- a/src/diffusers/modular_pipelines/wan_animate_2/encoders.py +++ b/src/diffusers/modular_pipelines/wan_animate_2/encoders.py @@ -81,11 +81,19 @@ def clip_visual_encode(image_encoder, tensor, device, dtype): return out.hidden_states[-2] -def get_i2v_mask(lat_t, lat_h, lat_w, mask_len=1, device="cuda"): +def get_i2v_mask(lat_t, lat_h, lat_w, mask_len=1, device=None): """Create an i2v mask in latent space. mask_len is in PIXEL space. Returns [4, lat_t, lat_h, lat_w] (no batch dim). + + Args: + device: device on which the mask is allocated. Must be passed explicitly by the caller. """ + if device is None: + raise ValueError( + "`device` must be specified when calling `get_i2v_mask`. It used to default to 'cuda', which " + "silently allocated the mask on CUDA and broke every non-CUDA accelerator (NPU/XPU/MPS/CPU)." + ) msk = torch.zeros(1, (lat_t - 1) * 4 + 1, lat_h, lat_w, device=device) msk[:, :mask_len] = 1 msk = torch.concat([torch.repeat_interleave(msk[:, 0:1], repeats=4, dim=1), msk[:, 1:]], dim=1)