launcher.py 6.29 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3

4
5
import asyncio
import signal
6
import socket
7
from http import HTTPStatus
8
from typing import Any
9
10

import uvicorn
11
from fastapi import FastAPI, Request, Response
12

13
from vllm import envs
14
from vllm.engine.protocol import EngineClient
15
16
17
18
from vllm.entrypoints.constants import (
    H11_MAX_HEADER_COUNT_DEFAULT,
    H11_MAX_INCOMPLETE_EVENT_SIZE_DEFAULT,
)
19
from vllm.entrypoints.ssl import SSLCertRefresher
20
from vllm.logger import init_logger
21
from vllm.utils.network_utils import find_process_using_port
22
from vllm.v1.engine.exceptions import EngineDeadError, EngineGenerateError
23
24
25
26

logger = init_logger(__name__)


27
28
async def serve_http(
    app: FastAPI,
29
    sock: socket.socket | None,
30
31
32
    enable_ssl_refresh: bool = False,
    **uvicorn_kwargs: Any,
):
33
34
35
36
37
    """
    Start a FastAPI app using Uvicorn, with support for custom Uvicorn config
    options.  Supports http header limits via h11_max_incomplete_event_size and
    h11_max_header_count.
    """
38
    logger.info("Available routes are:")
39
    # post endpoints
40
41
42
43
44
45
46
    for route in app.routes:
        methods = getattr(route, "methods", None)
        path = getattr(route, "path", None)

        if methods is None or path is None:
            continue

47
        logger.info("Route: %s, Methods: %s", path, ", ".join(methods))
48

49
50
51
52
53
54
55
56
57
58
59
    # other endpoints
    for route in app.routes:
        endpoint = getattr(route, "endpoint", None)
        methods = getattr(route, "methods", None)
        path = getattr(route, "path", None)

        if endpoint is None or path is None or methods is not None:
            continue

        logger.info("Route: %s, Endpoint: %s", path, endpoint.__name__)

60
61
    # Extract header limit options if present
    h11_max_incomplete_event_size = uvicorn_kwargs.pop(
62
63
        "h11_max_incomplete_event_size", None
    )
64
65
66
67
68
69
70
71
    h11_max_header_count = uvicorn_kwargs.pop("h11_max_header_count", None)

    # Set safe defaults if not provided
    if h11_max_incomplete_event_size is None:
        h11_max_incomplete_event_size = H11_MAX_INCOMPLETE_EVENT_SIZE_DEFAULT
    if h11_max_header_count is None:
        h11_max_header_count = H11_MAX_HEADER_COUNT_DEFAULT

72
    config = uvicorn.Config(app, **uvicorn_kwargs)
73
74
75
    # Set header limits
    config.h11_max_incomplete_event_size = h11_max_incomplete_event_size
    config.h11_max_header_count = h11_max_header_count
76
    config.load()
77
    server = uvicorn.Server(config)
78
    _add_shutdown_handlers(app, server)
79
80
81

    loop = asyncio.get_running_loop()

82
83
84
85
86
87
88
89
90
91
92
93
94
    watchdog_task = loop.create_task(watchdog_loop(server, app.state.engine_client))
    server_task = loop.create_task(server.serve(sockets=[sock] if sock else None))

    ssl_cert_refresher = (
        None
        if not enable_ssl_refresh
        else SSLCertRefresher(
            ssl_context=config.ssl,
            key_path=config.ssl_keyfile,
            cert_path=config.ssl_certfile,
            ca_path=config.ssl_ca_certs,
        )
    )
95

96
97
98
    def signal_handler() -> None:
        # prevents the uvicorn signal handler to exit early
        server_task.cancel()
99
        watchdog_task.cancel()
100
101
        if ssl_cert_refresher:
            ssl_cert_refresher.stop()
102
103
104
105
106
107
108
109
110
111
112

    async def dummy_shutdown() -> None:
        pass

    loop.add_signal_handler(signal.SIGINT, signal_handler)
    loop.add_signal_handler(signal.SIGTERM, signal_handler)

    try:
        await server_task
        return dummy_shutdown()
    except asyncio.CancelledError:
113
114
115
        port = uvicorn_kwargs["port"]
        process = find_process_using_port(port)
        if process is not None:
116
            logger.warning(
117
                "port %s is used by process %s launched with command:\n%s",
118
119
120
121
                port,
                process,
                " ".join(process.cmdline()),
            )
122
        logger.info("Shutting down FastAPI HTTP server.")
123
        return server.shutdown()
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
    finally:
        watchdog_task.cancel()


async def watchdog_loop(server: uvicorn.Server, engine: EngineClient):
    """
    # Watchdog task that runs in the background, checking
    # for error state in the engine. Needed to trigger shutdown
    # if an exception arises is StreamingResponse() generator.
    """
    VLLM_WATCHDOG_TIME_S = 5.0
    while True:
        await asyncio.sleep(VLLM_WATCHDOG_TIME_S)
        terminate_if_errored(server, engine)


def terminate_if_errored(server: uvicorn.Server, engine: EngineClient):
    """
    See discussions here on shutting down a uvicorn server
    https://github.com/encode/uvicorn/discussions/1103
    In this case we cannot await the server shutdown here
    because handler must first return to close the connection
    for this request.
    """
    engine_errored = engine.errored and not engine.is_running
    if not envs.VLLM_KEEP_ALIVE_ON_ENGINE_DEATH and engine_errored:
        server.should_exit = True
151
152


153
def _add_shutdown_handlers(app: FastAPI, server: uvicorn.Server) -> None:
154
155
156
    """
    VLLM V1 AsyncLLM catches exceptions and returns
    only two types: EngineGenerateError and EngineDeadError.
157

158
159
160
    EngineGenerateError is raised by the per request generate()
    method. This error could be request specific (and therefore
    recoverable - e.g. if there is an error in input processing).
161

162
163
    EngineDeadError is raised by the background output_handler
    method. This error is global and therefore not recoverable.
164

165
166
167
168
169
170
171
172
173
174
175
176
    We register these @app.exception_handlers to return nice
    responses to the end user if they occur and shut down if needed.
    See https://fastapi.tiangolo.com/tutorial/handling-errors/
    for more details on how exception handlers work.

    If an exception is encountered in a StreamingResponse
    generator, the exception is not raised, since we already sent
    a 200 status. Rather, we send an error message as the next chunk.
    Since the exception is not raised, this means that the server
    will not automatically shut down. Instead, we use the watchdog
    background task for check for errored state.
    """
177
178

    @app.exception_handler(RuntimeError)
179
180
181
182
183
184
185
    @app.exception_handler(EngineDeadError)
    @app.exception_handler(EngineGenerateError)
    async def runtime_exception_handler(request: Request, __):
        terminate_if_errored(
            server=server,
            engine=request.app.state.engine_client,
        )
186
187

        return Response(status_code=HTTPStatus.INTERNAL_SERVER_ERROR)