update

Merge branch 'main' into custom-device-map
add custom mesh support
2026-02-25 04:10:34 +08:00 · 2026-02-24 09:45:26 +01:00 · 2026-02-24 09:20:16 +01:00 · 2026-02-02 13:12:09 +05:30
8 changed files with 114 additions and 189 deletions
--- a/src/diffusers/models/_modeling_parallel.py
+++ b/src/diffusers/models/_modeling_parallel.py
@@ -60,6 +60,16 @@ class ContextParallelConfig:
        rotate_method (`str`, *optional*, defaults to `"allgather"`):
            Method to use for rotating key/value states across devices in ring attention. Currently, only `"allgather"`
            is supported.
+        ulysses_anything (`bool`, *optional*, defaults to `False`):
+            Whether to enable "Ulysses Anything" mode, which supports arbitrary sequence lengths and head counts that
+            are not evenly divisible by `ulysses_degree`. When enabled, `ulysses_degree` must be greater than 1 and
+            `ring_degree` must be 1.
+        mesh (`torch.distributed.device_mesh.DeviceMesh`, *optional*):
+            A custom device mesh to use for context parallelism. If provided, this mesh will be used instead of
+            creating a new one. This is useful when combining context parallelism with other parallelism strategies
+            (e.g., FSDP, tensor parallelism) that share the same device mesh. The mesh must have both "ring" and
+            "ulysses" dimensions. Use size 1 for dimensions not being used (e.g., `mesh_shape=(2, 1, 4)` with
+            `mesh_dim_names=("ring", "ulysses", "fsdp")` for ring attention only with FSDP).

    """

@@ -68,6 +78,7 @@ class ContextParallelConfig:
    convert_to_fp32: bool = True
    # TODO: support alltoall
    rotate_method: Literal["allgather", "alltoall"] = "allgather"
+    mesh: torch.distributed.device_mesh.DeviceMesh | None = None
    # Whether to enable ulysses anything attention to support
    # any sequence lengths and any head numbers.
    ulysses_anything: bool = False
@@ -124,7 +135,7 @@ class ContextParallelConfig:
                f"The product of `ring_degree` ({self.ring_degree}) and `ulysses_degree` ({self.ulysses_degree}) must not exceed the world size ({world_size})."
            )

-        self._flattened_mesh = self._mesh._flatten()
+        self._flattened_mesh = self._mesh["ring", "ulysses"]._flatten()
        self._ring_mesh = self._mesh["ring"]
        self._ulysses_mesh = self._mesh["ulysses"]
        self._ring_local_rank = self._ring_mesh.get_local_rank()
--- a/src/diffusers/models/attention_dispatch.py
+++ b/src/diffusers/models/attention_dispatch.py
@@ -62,8 +62,6 @@ _REQUIRED_FLEX_VERSION = "2.5.0"
 _REQUIRED_XLA_VERSION = "2.2"
 _REQUIRED_XFORMERS_VERSION = "0.0.29"

-logger = get_logger(__name__)  # pylint: disable=invalid-name
-
 _CAN_USE_FLASH_ATTN = is_flash_attn_available() and is_flash_attn_version(">=", _REQUIRED_FLASH_VERSION)
 _CAN_USE_FLASH_ATTN_3 = is_flash_attn_3_available()
 _CAN_USE_AITER_ATTN = is_aiter_available() and is_aiter_version(">=", _REQUIRED_AITER_VERSION)
@@ -75,18 +73,8 @@ _CAN_USE_XFORMERS_ATTN = is_xformers_available() and is_xformers_version(">=", _


 if _CAN_USE_FLASH_ATTN:
-    try:
-        from flash_attn import flash_attn_func, flash_attn_varlen_func
-        from flash_attn.flash_attn_interface import _wrapped_flash_attn_backward, _wrapped_flash_attn_forward
-    except (ImportError, OSError, RuntimeError) as e:
-        # Handle ABI mismatch or other import failures gracefully.
-        # This can happen when flash_attn was compiled against a different PyTorch version.
-        logger.warning(f"flash_attn is installed but failed to import: {e}. Falling back to native PyTorch attention.")
-        _CAN_USE_FLASH_ATTN = False
-        flash_attn_func = None
-        flash_attn_varlen_func = None
-        _wrapped_flash_attn_backward = None
-        _wrapped_flash_attn_forward = None
+    from flash_attn import flash_attn_func, flash_attn_varlen_func
+    from flash_attn.flash_attn_interface import _wrapped_flash_attn_backward, _wrapped_flash_attn_forward
 else:
    flash_attn_func = None
    flash_attn_varlen_func = None
@@ -95,47 +83,26 @@ else:


 if _CAN_USE_FLASH_ATTN_3:
-    try:
-        from flash_attn_interface import flash_attn_func as flash_attn_3_func
-        from flash_attn_interface import flash_attn_varlen_func as flash_attn_3_varlen_func
-    except (ImportError, OSError, RuntimeError) as e:
-        logger.warning(f"flash_attn_3 failed to import: {e}. Falling back to native attention.")
-        _CAN_USE_FLASH_ATTN_3 = False
-        flash_attn_3_func = None
-        flash_attn_3_varlen_func = None
+    from flash_attn_interface import flash_attn_func as flash_attn_3_func
+    from flash_attn_interface import flash_attn_varlen_func as flash_attn_3_varlen_func
 else:
    flash_attn_3_func = None
    flash_attn_3_varlen_func = None

 if _CAN_USE_AITER_ATTN:
-    try:
-        from aiter import flash_attn_func as aiter_flash_attn_func
-    except (ImportError, OSError, RuntimeError) as e:
-        logger.warning(f"aiter failed to import: {e}. Falling back to native attention.")
-        _CAN_USE_AITER_ATTN = False
-        aiter_flash_attn_func = None
+    from aiter import flash_attn_func as aiter_flash_attn_func
 else:
    aiter_flash_attn_func = None

 if _CAN_USE_SAGE_ATTN:
-    try:
-        from sageattention import (
-            sageattn,
-            sageattn_qk_int8_pv_fp8_cuda,
-            sageattn_qk_int8_pv_fp8_cuda_sm90,
-            sageattn_qk_int8_pv_fp16_cuda,
-            sageattn_qk_int8_pv_fp16_triton,
-            sageattn_varlen,
-        )
-    except (ImportError, OSError, RuntimeError) as e:
-        logger.warning(f"sageattention failed to import: {e}. Falling back to native attention.")
-        _CAN_USE_SAGE_ATTN = False
-        sageattn = None
-        sageattn_qk_int8_pv_fp8_cuda = None
-        sageattn_qk_int8_pv_fp8_cuda_sm90 = None
-        sageattn_qk_int8_pv_fp16_cuda = None
-        sageattn_qk_int8_pv_fp16_triton = None
-        sageattn_varlen = None
+    from sageattention import (
+        sageattn,
+        sageattn_qk_int8_pv_fp8_cuda,
+        sageattn_qk_int8_pv_fp8_cuda_sm90,
+        sageattn_qk_int8_pv_fp16_cuda,
+        sageattn_qk_int8_pv_fp16_triton,
+        sageattn_varlen,
+    )
 else:
    sageattn = None
    sageattn_qk_int8_pv_fp16_cuda = None
@@ -146,48 +113,26 @@ else:


 if _CAN_USE_FLEX_ATTN:
-    try:
-        # We cannot import the flex_attention function from the package directly because it is expected (from the
-        # pytorch documentation) that the user may compile it. If we import directly, we will not have access to the
-        # compiled function.
-        import torch.nn.attention.flex_attention as flex_attention
-    except (ImportError, OSError, RuntimeError) as e:
-        logger.warning(f"flex_attention failed to import: {e}. Falling back to native attention.")
-        _CAN_USE_FLEX_ATTN = False
-        flex_attention = None
-else:
-    flex_attention = None
+    # We cannot import the flex_attention function from the package directly because it is expected (from the
+    # pytorch documentation) that the user may compile it. If we import directly, we will not have access to the
+    # compiled function.
+    import torch.nn.attention.flex_attention as flex_attention


 if _CAN_USE_NPU_ATTN:
-    try:
-        from torch_npu import npu_fusion_attention
-    except (ImportError, OSError, RuntimeError) as e:
-        logger.warning(f"torch_npu failed to import: {e}. Falling back to native attention.")
-        _CAN_USE_NPU_ATTN = False
-        npu_fusion_attention = None
+    from torch_npu import npu_fusion_attention
 else:
    npu_fusion_attention = None


 if _CAN_USE_XLA_ATTN:
-    try:
-        from torch_xla.experimental.custom_kernel import flash_attention as xla_flash_attention
-    except (ImportError, OSError, RuntimeError) as e:
-        logger.warning(f"torch_xla failed to import: {e}. Falling back to native attention.")
-        _CAN_USE_XLA_ATTN = False
-        xla_flash_attention = None
+    from torch_xla.experimental.custom_kernel import flash_attention as xla_flash_attention
 else:
    xla_flash_attention = None


 if _CAN_USE_XFORMERS_ATTN:
-    try:
-        import xformers.ops as xops
-    except (ImportError, OSError, RuntimeError) as e:
-        logger.warning(f"xformers failed to import: {e}. Falling back to native attention.")
-        _CAN_USE_XFORMERS_ATTN = False
-        xops = None
+    import xformers.ops as xops
 else:
    xops = None

@@ -213,6 +158,8 @@ else:
    _register_fake = register_fake_no_op


+logger = get_logger(__name__)  # pylint: disable=invalid-name
+
 # TODO(aryan): Add support for the following:
 # - Sage Attention++
 # - block sparse, radial and other attention methods
@@ -1865,12 +1812,9 @@ class TemplatedRingAttention(torch.autograd.Function):
                out = out.to(torch.float32)
                lse = lse.to(torch.float32)

-            # lse must be 4-D to broadcast with out (B, S, H, D).
-            # Some backends (e.g. cuDNN on torch>=2.9) already return a
-            # trailing-1 dim; others (e.g. flash-hub / native-flash) always
-            # return 3-D lse, so we add the dim here when needed.
-            # See: https://github.com/huggingface/diffusers/pull/12693#issuecomment-3627519544
-            if lse.ndim == 3:
+            # Refer to:
+            # https://github.com/huggingface/diffusers/pull/12693#issuecomment-3627519544
+            if is_torch_version("<", "2.9.0"):
                lse = lse.unsqueeze(-1)
            if prev_out is not None:
                out = prev_out - torch.nn.functional.sigmoid(lse - prev_lse) * (prev_out - out)
@@ -2157,11 +2101,10 @@ def _templated_unified_attention(
        scatter_idx,
    )
    if return_lse:
-        # lse from TemplatedRingAttention is 3-D (B, S, H_LOCAL) after its
-        # final squeeze(-1). SeqAllToAllDim requires a 4-D input, so we add
-        # the trailing dim here and remove it after the collective.
-        # See: https://github.com/huggingface/diffusers/pull/12693#issuecomment-3627519544
-        if lse.ndim == 3:
+        # lse is of shape (B, S, H_LOCAL, 1)
+        # Refer to:
+        # https://github.com/huggingface/diffusers/pull/12693#issuecomment-3627519544
+        if is_torch_version("<", "2.9.0"):
            lse = lse.unsqueeze(-1)  # (B, S, H_LOCAL, 1)
        lse = SeqAllToAllDim.apply(ulysses_group, lse, gather_idx, scatter_idx)
        lse = lse.squeeze(-1)
--- a/src/diffusers/models/modeling_utils.py
+++ b/src/diffusers/models/modeling_utils.py
@@ -1567,7 +1567,7 @@ class ModelMixin(torch.nn.Module, PushToHubMixin):
        mesh = None
        if config.context_parallel_config is not None:
            cp_config = config.context_parallel_config
-            mesh = torch.distributed.device_mesh.init_device_mesh(
+            mesh = cp_config.mesh or torch.distributed.device_mesh.init_device_mesh(
                device_type=device_type,
                mesh_shape=cp_config.mesh_shape,
                mesh_dim_names=cp_config.mesh_dim_names,
--- a/tests/models/testing_utils/init.py
+++ b/tests/models/testing_utils/init.py
@@ -13,7 +13,7 @@ from .compile import TorchCompileTesterMixin
 from .ip_adapter import IPAdapterTesterMixin
 from .lora import LoraHotSwappingForModelTesterMixin, LoraTesterMixin
 from .memory import CPUOffloadTesterMixin, GroupOffloadTesterMixin, LayerwiseCastingTesterMixin, MemoryTesterMixin
-from .parallelism import ContextParallelAttentionBackendsTesterMixin, ContextParallelTesterMixin
+from .parallelism import ContextParallelTesterMixin
 from .quantization import (
    BitsAndBytesCompileTesterMixin,
    BitsAndBytesConfigMixin,
@@ -45,7 +45,6 @@ __all__ = [
    "BitsAndBytesTesterMixin",
    "CacheTesterMixin",
    "ContextParallelTesterMixin",
-    "ContextParallelAttentionBackendsTesterMixin",
    "CPUOffloadTesterMixin",
    "FasterCacheConfigMixin",
    "FasterCacheTesterMixin",
--- a/tests/models/testing_utils/parallelism.py
+++ b/tests/models/testing_utils/parallelism.py
@@ -23,8 +23,10 @@ import torch.multiprocessing as mp

 from diffusers.models._modeling_parallel import ContextParallelConfig

-from ...testing_utils import is_context_parallel, is_kernels_available, require_torch_multi_accelerator
-from .utils import _maybe_cast_to_bf16
+from ...testing_utils import (
+    is_context_parallel,
+    require_torch_multi_accelerator,
+)


 def _find_free_port():
@@ -36,9 +38,7 @@ def _find_free_port():
    return port


-def _context_parallel_worker(
-    rank, world_size, master_port, model_class, init_dict, cp_dict, inputs_dict, return_dict, attention_backend=None
-):
+def _context_parallel_worker(rank, world_size, master_port, model_class, init_dict, cp_dict, inputs_dict, return_dict):
    """Worker function for context parallel testing."""
    try:
        # Set up distributed environment
@@ -59,23 +59,8 @@ def _context_parallel_worker(
        model.to(device)
        model.eval()

-        # Cast as needed.
-        model, inputs_dict = _maybe_cast_to_bf16(attention_backend, model, inputs_dict)
-
        # Move inputs to device
-        inputs_on_device = {}
-        for key, value in inputs_dict.items():
-            if isinstance(value, torch.Tensor):
-                inputs_on_device[key] = value.to(device)
-            else:
-                inputs_on_device[key] = value
-
-        # Enable attention backend
-        if attention_backend:
-            try:
-                model.set_attention_backend(attention_backend)
-            except Exception as e:
-                pytest.skip(f"Skipping test because of exception: {e}.")
+        inputs_on_device = {k: v.to(device) if isinstance(v, torch.Tensor) else v for k, v in inputs_dict.items()}

        # Enable context parallelism
        cp_config = ContextParallelConfig(**cp_dict)
@@ -99,6 +84,59 @@ def _context_parallel_worker(
            dist.destroy_process_group()


+def _custom_mesh_worker(
+    rank,
+    world_size,
+    master_port,
+    model_class,
+    init_dict,
+    cp_dict,
+    mesh_shape,
+    mesh_dim_names,
+    inputs_dict,
+    return_dict,
+):
+    """Worker function for context parallel testing with a user-provided custom DeviceMesh."""
+    try:
+        os.environ["MASTER_ADDR"] = "localhost"
+        os.environ["MASTER_PORT"] = str(master_port)
+        os.environ["RANK"] = str(rank)
+        os.environ["WORLD_SIZE"] = str(world_size)
+
+        dist.init_process_group(backend="nccl", rank=rank, world_size=world_size)
+
+        torch.cuda.set_device(rank)
+        device = torch.device(f"cuda:{rank}")
+
+        model = model_class(**init_dict)
+        model.to(device)
+        model.eval()
+
+        inputs_on_device = {k: v.to(device) if isinstance(v, torch.Tensor) else v for k, v in inputs_dict.items()}
+
+        # DeviceMesh must be created after init_process_group, inside each worker process.
+        mesh = torch.distributed.device_mesh.init_device_mesh(
+            "cuda", mesh_shape=mesh_shape, mesh_dim_names=mesh_dim_names
+        )
+        cp_config = ContextParallelConfig(**cp_dict, mesh=mesh)
+        model.enable_parallelism(config=cp_config)
+
+        with torch.no_grad():
+            output = model(**inputs_on_device, return_dict=False)[0]
+
+        if rank == 0:
+            return_dict["status"] = "success"
+            return_dict["output_shape"] = list(output.shape)
+
+    except Exception as e:
+        if rank == 0:
+            return_dict["status"] = "error"
+            return_dict["error"] = str(e)
+    finally:
+        if dist.is_initialized():
+            dist.destroy_process_group()
+
+
@is_context_parallel
@require_torch_multi_accelerator
 class ContextParallelTesterMixin:
@@ -137,75 +175,47 @@ class ContextParallelTesterMixin:
            f"Context parallel inference failed: {return_dict.get('error', 'Unknown error')}"
        )

-
-@is_context_parallel
-@require_torch_multi_accelerator
-class ContextParallelAttentionBackendsTesterMixin:
-    @pytest.mark.parametrize("cp_type", ["ulysses_degree", "ring_degree"])
    @pytest.mark.parametrize(
-        "attention_backend",
+        "cp_type,mesh_shape,mesh_dim_names",
        [
-            "native",
-            pytest.param(
-                "flash_hub",
-                marks=pytest.mark.skipif(not is_kernels_available(), reason="`kernels` is not available."),
-            ),
-            pytest.param(
-                "_flash_3_hub",
-                marks=pytest.mark.skipif(not is_kernels_available(), reason="`kernels` is not available."),
-            ),
+            ("ring_degree", (2, 1, 1), ("ring", "ulysses", "fsdp")),
+            ("ulysses_degree", (1, 2, 1), ("ring", "ulysses", "fsdp")),
        ],
+        ids=["ring-3d-fsdp", "ulysses-3d-fsdp"],
    )
-    @pytest.mark.parametrize("ulysses_anything", [True, False])
-    @torch.no_grad()
-    def test_context_parallel_attn_backend_inference(self, cp_type, attention_backend, ulysses_anything):
+    def test_context_parallel_custom_mesh(self, cp_type, mesh_shape, mesh_dim_names):
        if not torch.distributed.is_available():
            pytest.skip("torch.distributed is not available.")

-        if getattr(self.model_class, "_cp_plan", None) is None:
+        if not hasattr(self.model_class, "_cp_plan") or self.model_class._cp_plan is None:
            pytest.skip("Model does not have a _cp_plan defined for context parallel inference.")

-        if cp_type == "ring_degree":
-            if attention_backend == "native":
-                pytest.skip("Skipping test because ulysses isn't supported with native attention backend.")
-
-        if ulysses_anything and "ulysses" not in cp_type:
-            pytest.skip("Skipping test as ulysses anything needs the ulysses degree set.")
-
        world_size = 2
        init_dict = self.get_init_dict()
-        inputs_dict = self.get_dummy_inputs()
-
-        # Move all tensors to CPU for multiprocessing
-        inputs_dict = {k: v.cpu() if isinstance(v, torch.Tensor) else v for k, v in inputs_dict.items()}
+        inputs_dict = {k: v.cpu() if isinstance(v, torch.Tensor) else v for k, v in self.get_dummy_inputs().items()}
        cp_dict = {cp_type: world_size}
-        if ulysses_anything:
-            cp_dict.update({"ulysses_anything": ulysses_anything})

-        # Find a free port for distributed communication
        master_port = _find_free_port()
-
-        # Use multiprocessing manager for cross-process communication
        manager = mp.Manager()
        return_dict = manager.dict()

-        # Spawn worker processes
        mp.spawn(
-            _context_parallel_worker,
+            _custom_mesh_worker,
            args=(
                world_size,
                master_port,
                self.model_class,
                init_dict,
                cp_dict,
+                mesh_shape,
+                mesh_dim_names,
                inputs_dict,
                return_dict,
-                attention_backend,
            ),
            nprocs=world_size,
            join=True,
        )

        assert return_dict.get("status") == "success", (
-            f"Context parallel inference failed: {return_dict.get('error', 'Unknown error')}"
+            f"Custom mesh context parallel inference failed: {return_dict.get('error', 'Unknown error')}"
        )
--- a/tests/models/testing_utils/utils.py
+++ b/tests/models/testing_utils/utils.py
@@ -1,22 +0,0 @@
-import torch
-
-from diffusers.models.attention_dispatch import AttentionBackendName
-
-
-_BF16_REQUIRED_BACKENDS = {
-    AttentionBackendName._NATIVE_CUDNN,
-    AttentionBackendName.FLASH_HUB,
-    AttentionBackendName._FLASH_3_HUB,
-}
-
-
-def _maybe_cast_to_bf16(backend, model, inputs_dict):
-    """Cast model and floating-point inputs to bfloat16 when the backend requires it."""
-    if not backend or backend not in _BF16_REQUIRED_BACKENDS:
-        return model, inputs_dict
-    model = model.to(dtype=torch.bfloat16)
-    inputs_dict = {
-        k: v.to(dtype=torch.bfloat16) if isinstance(v, torch.Tensor) and v.is_floating_point() else v
-        for k, v in inputs_dict.items()
-    }
-    return model, inputs_dict
--- a/tests/models/transformers/test_models_transformer_flux.py
+++ b/tests/models/transformers/test_models_transformer_flux.py
@@ -29,7 +29,6 @@ from ..testing_utils import (
    BaseModelTesterConfig,
    BitsAndBytesCompileTesterMixin,
    BitsAndBytesTesterMixin,
-    ContextParallelAttentionBackendsTesterMixin,
    ContextParallelTesterMixin,
    FasterCacheTesterMixin,
    FirstBlockCacheTesterMixin,
@@ -229,12 +228,6 @@ class TestFluxTransformerContextParallel(FluxTransformerTesterConfig, ContextPar
    """Context Parallel inference tests for Flux Transformer"""


-class TestFluxTransformerContextParallelAttnBackends(
-    FluxTransformerTesterConfig, ContextParallelAttentionBackendsTesterMixin
-):
-    """Context Parallel inference x attention backends tests for Flux Transformer"""
-
-
 class TestFluxTransformerIPAdapter(FluxTransformerTesterConfig, IPAdapterTesterMixin):
    """IP Adapter tests for Flux Transformer."""

--- a/utils/generate_model_tests.py
+++ b/utils/generate_model_tests.py
@@ -72,7 +72,6 @@ OPTIONAL_TESTERS = [
    # Other testers
    ("SingleFileTesterMixin", "single_file"),
    ("IPAdapterTesterMixin", "ip_adapter"),
-    ("ContextParallelAttentionBackendsTesterMixin", "cp_attn"),
 ]


@@ -230,14 +229,7 @@ def determine_testers(model_info: dict, include_optional: list[str], imports: se

    for tester, flag in OPTIONAL_TESTERS:
        if flag in include_optional:
-            if tester == "ContextParallelAttentionBackendsTesterMixin":
-                if (
-                    "cp_attn" in include_optional
-                    and "_cp_plan" in model_info["attributes"]
-                    and model_info["attributes"]["_cp_plan"] is not None
-                ):
-                    testers.append(tester)
-            elif tester not in testers:
+            if tester not in testers:
                testers.append(tester)

    return testers
@@ -538,7 +530,6 @@ def main():
            "faster_cache",
            "single_file",
            "ip_adapter",
-            "cp_attn",
            "all",
        ],
        help="Optional testers to include",
Author	SHA1	Message	Date
Dhruv Nair	96f97f8214	update	2026-02-24 09:45:26 +01:00
Dhruv Nair	f8b95ff263	Merge branch 'main' into custom-device-map	2026-02-24 09:20:16 +01:00
DN6	cfff46069d	add custom mesh support	2026-02-02 13:12:09 +05:30