test_serving_models.py 5.2 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3

4
5
6
7
8
9
from http import HTTPStatus
from unittest.mock import MagicMock

import pytest

from vllm.config import ModelConfig
10
from vllm.engine.protocol import EngineClient
11
from vllm.entrypoints.openai.protocol import (ErrorResponse,
12
13
                                              LoadLoRAAdapterRequest,
                                              UnloadLoRAAdapterRequest)
14
15
from vllm.entrypoints.openai.serving_models import (BaseModelPath,
                                                    OpenAIServingModels)
16
from vllm.lora.request import LoRARequest
17

18
MODEL_NAME = "meta-llama/Llama-3.2-1B-Instruct"
19
BASE_MODEL_PATHS = [BaseModelPath(name=MODEL_NAME, model_path=MODEL_NAME)]
20
21
22
23
24
25
LORA_LOADING_SUCCESS_MESSAGE = (
    "Success: LoRA adapter '{lora_name}' added successfully.")
LORA_UNLOADING_SUCCESS_MESSAGE = (
    "Success: LoRA adapter '{lora_name}' removed successfully.")


26
async def _async_serving_models_init() -> OpenAIServingModels:
27
    mock_model_config = MagicMock(spec=ModelConfig)
28
    mock_engine_client = MagicMock(spec=EngineClient)
29
30
31
    # Set the max_model_len attribute to avoid missing attribute
    mock_model_config.max_model_len = 2048

32
33
    serving_models = OpenAIServingModels(engine_client=mock_engine_client,
                                         base_model_paths=BASE_MODEL_PATHS,
34
35
36
                                         model_config=mock_model_config,
                                         lora_modules=None,
                                         prompt_adapters=None)
37
    await serving_models.init_static_loras()
38
39

    return serving_models
40
41


42
43
@pytest.mark.asyncio
async def test_serving_model_name():
44
45
    serving_models = await _async_serving_models_init()
    assert serving_models.model_name(None) == MODEL_NAME
46
47
48
    request = LoRARequest(lora_name="adapter",
                          lora_path="/path/to/adapter2",
                          lora_int_id=1)
49
    assert serving_models.model_name(request) == request.lora_name
50
51


52
53
@pytest.mark.asyncio
async def test_load_lora_adapter_success():
54
    serving_models = await _async_serving_models_init()
55
    request = LoadLoRAAdapterRequest(lora_name="adapter",
56
                                     lora_path="/path/to/adapter2")
57
    response = await serving_models.load_lora_adapter(request)
58
    assert response == LORA_LOADING_SUCCESS_MESSAGE.format(lora_name='adapter')
59
    assert len(serving_models.lora_requests) == 1
60
61
    assert "adapter" in serving_models.lora_requests
    assert serving_models.lora_requests["adapter"].lora_name == "adapter"
62
63
64
65


@pytest.mark.asyncio
async def test_load_lora_adapter_missing_fields():
66
    serving_models = await _async_serving_models_init()
67
    request = LoadLoRAAdapterRequest(lora_name="", lora_path="")
68
    response = await serving_models.load_lora_adapter(request)
69
70
71
72
73
74
75
    assert isinstance(response, ErrorResponse)
    assert response.type == "InvalidUserInput"
    assert response.code == HTTPStatus.BAD_REQUEST


@pytest.mark.asyncio
async def test_load_lora_adapter_duplicate():
76
    serving_models = await _async_serving_models_init()
77
    request = LoadLoRAAdapterRequest(lora_name="adapter1",
78
                                     lora_path="/path/to/adapter1")
79
    response = await serving_models.load_lora_adapter(request)
80
81
    assert response == LORA_LOADING_SUCCESS_MESSAGE.format(
        lora_name='adapter1')
82
    assert len(serving_models.lora_requests) == 1
83

84
    request = LoadLoRAAdapterRequest(lora_name="adapter1",
85
                                     lora_path="/path/to/adapter1")
86
    response = await serving_models.load_lora_adapter(request)
87
88
89
    assert isinstance(response, ErrorResponse)
    assert response.type == "InvalidUserInput"
    assert response.code == HTTPStatus.BAD_REQUEST
90
    assert len(serving_models.lora_requests) == 1
91
92
93
94


@pytest.mark.asyncio
async def test_unload_lora_adapter_success():
95
    serving_models = await _async_serving_models_init()
96
    request = LoadLoRAAdapterRequest(lora_name="adapter1",
97
                                     lora_path="/path/to/adapter1")
98
99
    response = await serving_models.load_lora_adapter(request)
    assert len(serving_models.lora_requests) == 1
100

101
    request = UnloadLoRAAdapterRequest(lora_name="adapter1")
102
    response = await serving_models.unload_lora_adapter(request)
103
104
    assert response == LORA_UNLOADING_SUCCESS_MESSAGE.format(
        lora_name='adapter1')
105
    assert len(serving_models.lora_requests) == 0
106
107
108
109


@pytest.mark.asyncio
async def test_unload_lora_adapter_missing_fields():
110
    serving_models = await _async_serving_models_init()
111
    request = UnloadLoRAAdapterRequest(lora_name="", lora_int_id=None)
112
    response = await serving_models.unload_lora_adapter(request)
113
114
115
116
117
118
119
    assert isinstance(response, ErrorResponse)
    assert response.type == "InvalidUserInput"
    assert response.code == HTTPStatus.BAD_REQUEST


@pytest.mark.asyncio
async def test_unload_lora_adapter_not_found():
120
    serving_models = await _async_serving_models_init()
121
    request = UnloadLoRAAdapterRequest(lora_name="nonexistent_adapter")
122
    response = await serving_models.unload_lora_adapter(request)
123
    assert isinstance(response, ErrorResponse)
124
125
    assert response.type == "NotFoundError"
    assert response.code == HTTPStatus.NOT_FOUND