fa_utils.py 4.13 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3
4

from vllm.logger import init_logger
5
from vllm.platforms import current_platform
6
import torch
7
8
9

logger = init_logger(__name__)

10
if current_platform.is_cuda():
11
    from vllm._custom_ops import reshape_and_cache_flash
12
13
14
15
    from vllm.vllm_flash_attn import (  # type: ignore[attr-defined]
        flash_attn_varlen_func,
        get_scheduler_metadata,
    )
16
elif current_platform.is_xpu():
17
    from vllm._ipex_ops import ipex_ops
18

19
20
21
    reshape_and_cache_flash = ipex_ops.reshape_and_cache_flash
    flash_attn_varlen_func = ipex_ops.flash_attn_varlen_func
    get_scheduler_metadata = ipex_ops.get_scheduler_metadata
22

23
24
elif current_platform.is_rocm():
    try:
25
        from vllm._custom_ops import reshape_and_cache_cuda
26
        from flash_attn import flash_attn_varlen_func, vllm_flash_attn_varlen_func
27
28
29
30
31
    except ImportError as e:
        raise ImportError(
            "Rocm platform requires upstream flash-attn "
            "to be installed. Please install flash-attn first."
        ) from e
32
    
33

34
def get_flash_attn_version(requires_alibi: bool = False) -> int | None:
35
36
    # import here to avoid circular dependencies
    from vllm.platforms import current_platform
37

38
39
    if current_platform.is_xpu():
        return 2
40
41
    if current_platform.is_rocm():
        # ROCm doesn't use vllm_flash_attn; return None to skip fa_version arg
42
        return 2 # None
43
44
    try:
        from vllm.vllm_flash_attn.flash_attn_interface import (
45
46
47
48
            fa_version_unsupported_reason,
            is_fa_version_supported,
        )

49
50
51
52
53
        device_capability = current_platform.get_device_capability()

        assert device_capability is not None

        # 1. default version depending on platform
54
55
56
        fa_version = (
            3 if (device_capability.major == 9 and is_fa_version_supported(3)) else 2
        )
57

58
        # 2. override if passed by environment or config
59
        from vllm.config import get_current_vllm_config_or_none
60

61
62
63
64
65
        vllm_config = get_current_vllm_config_or_none()
        if (
            vllm_config is not None
            and vllm_config.attention_config.flash_attn_version is not None
        ):
66
            fa_version = vllm_config.attention_config.flash_attn_version
67
68
69

        # 3. fallback for unsupported combinations
        if device_capability.major == 10 and fa_version == 3:
70
            logger.warning_once(
71
                "Cannot use FA version 3 on Blackwell platform, "
72
73
                "defaulting to FA version 2."
            )
74
75
76
            fa_version = 2

        if requires_alibi and fa_version == 3:
77
78
79
            logger.warning_once(
                "Cannot use FA version 3 with ALiBi, defaulting to FA version 2."
            )
80
81
82
            fa_version = 2

        if not is_fa_version_supported(fa_version):
83
84
85
86
87
            logger.error(
                "Cannot use FA version %d is not supported due to %s",
                fa_version,
                fa_version_unsupported_reason(fa_version),
            )
88
89
90
91
92

        assert is_fa_version_supported(fa_version)
        return fa_version
    except (ImportError, AssertionError):
        return None
93
94
95


def flash_attn_supports_fp8() -> bool:
96
97
    if torch.cuda.get_device_properties("cuda").gcnArchName.split(':')[0] == "gfx938":
        return True
98
99
    return (
        get_flash_attn_version() == 3
100
        and current_platform.is_device_capability_family(90)
101
    )
102
103


104
105
106
107
108
def flash_attn_supports_sinks() -> bool:
    if current_platform.is_xpu():
        return True
    else:
        return get_flash_attn_version() == 3
109
110


111
112
def flash_attn_supports_mla():
    from vllm.platforms import current_platform
113

114
115
116
    if current_platform.is_cuda():
        try:
            from vllm.vllm_flash_attn.flash_attn_interface import (
117
118
119
                is_fa_version_supported,
            )

120
121
122
            return is_fa_version_supported(
                3
            ) and current_platform.is_device_capability_family(90)
123
124
125
126
127
        except (ImportError, AssertionError):
            pass
    return False


128
def is_flash_attn_varlen_func_available() -> bool:
zhuwenwen's avatar
zhuwenwen committed
129
    return current_platform.is_cuda() or current_platform.is_rocm() or current_platform.is_xpu()