"vllm/vscode:/vscode.git/clone" did not exist on "54951ac4bfb7f4224cb8f5ffc89b214c950107d8"
test_cascade_attention.py 1.2 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3

4
5
import pytest

6
7
from vllm import LLM, SamplingParams

8
9
from ...utils import fork_new_process_for_each_test

10

11
12
13
14
@fork_new_process_for_each_test
@pytest.mark.parametrize("attn_backend",
                         ["FLASH_ATTN_VLLM_V1", "FLASHINFER_VLLM_V1"])
def test_cascade_attention(example_system_message, monkeypatch, attn_backend):
15
16
17
18
    prompt = "\n<User>: Implement fibonacci sequence in Python.\n<Claude>:"

    with monkeypatch.context() as m:
        m.setenv("VLLM_USE_V1", "1")
19
        m.setenv("VLLM_ATTENTION_BACKEND", attn_backend)
20
21
22
23
24
25
26
27
28
29
30
31
32
33

        llm = LLM(model="Qwen/Qwen2-1.5B-Instruct")
        sampling_params = SamplingParams(temperature=0.0, max_tokens=100)

        # No cascade attention.
        single_prompt = [example_system_message + prompt]
        responses = llm.generate(single_prompt, sampling_params)
        ref_output = responses[0].outputs[0].text

        # (Probably) Use cascade attention.
        prompts = [example_system_message + prompt] * 64
        responses = llm.generate(prompts, sampling_params)
        for response in responses:
            assert response.outputs[0].text == ref_output