test_serving_throughput.py 3.62 KB
Newer Older
1
import os
2
3
4
5
import unittest
from types import SimpleNamespace

from sglang.bench_serving import run_benchmark
Lianmin Zheng's avatar
Lianmin Zheng committed
6
from sglang.srt.server_args import ServerArgs
7
from sglang.srt.utils import kill_child_process
Yineng Zhang's avatar
Yineng Zhang committed
8
9
from sglang.test.test_utils import (
    DEFAULT_MODEL_NAME_FOR_TEST,
10
11
    DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
    DEFAULT_URL_FOR_TEST,
Yineng Zhang's avatar
Yineng Zhang committed
12
13
    popen_launch_server,
)
14
15
16
17
18
19
20
21
22
23
24
25
26


class TestServingThroughput(unittest.TestCase):
    def run_test(self, disable_radix_cache, disable_flashinfer, chunked_prefill_size):
        # Launch the server
        other_args = []
        if disable_radix_cache:
            other_args.append("--disable-radix-cache")
        if disable_flashinfer:
            other_args.append("--disable-flashinfer")
        other_args.extend(["--chunked-prefill-size", str(chunked_prefill_size)])

        model = DEFAULT_MODEL_NAME_FOR_TEST
27
        base_url = DEFAULT_URL_FOR_TEST
28
        process = popen_launch_server(
29
30
31
32
            model,
            base_url,
            timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
            other_args=other_args,
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
        )

        # Run benchmark
        num_prompts = 400
        args = SimpleNamespace(
            backend="sglang",
            base_url=base_url,
            host=None,
            port=None,
            dataset_name="random",
            dataset_path="",
            model=None,
            tokenizer=None,
            num_prompts=num_prompts,
            sharegpt_output_len=None,
            random_input_len=4096,
            random_output_len=2048,
            random_range_ratio=0.0,
            request_rate=float("inf"),
            multi=None,
            seed=0,
            output_file=None,
            disable_tqdm=False,
            disable_stream=False,
            disable_ignore_eos=False,
            extra_request_body=None,
        )

        try:
            res = run_benchmark(args)
        finally:
            kill_child_process(process.pid)

        assert res["completed"] == num_prompts
67
        return res
68
69

    def test_default(self):
70
        res = self.run_test(
Lianmin Zheng's avatar
Lianmin Zheng committed
71
72
73
            disable_radix_cache=ServerArgs.disable_radix_cache,
            disable_flashinfer=ServerArgs.disable_flashinfer,
            chunked_prefill_size=ServerArgs.chunked_prefill_size,
74
75
        )

76
        if os.getenv("SGLANG_IS_IN_CI", "false") == "true":
77
78
            # A100 (PCIE): 1450, H100 (SMX): 2550
            assert res["output_throughput"] > 2500
79

80
    def test_default_without_radix_cache(self):
81
        res = self.run_test(
82
            disable_radix_cache=True,
Lianmin Zheng's avatar
Lianmin Zheng committed
83
84
            disable_flashinfer=ServerArgs.disable_flashinfer,
            chunked_prefill_size=ServerArgs.chunked_prefill_size,
85
86
        )

87
        if os.getenv("SGLANG_IS_IN_CI", "false") == "true":
88
89
            # A100 (PCIE): 1500, H100 (SMX): 2850
            assert res["output_throughput"] > 2800
90

91
    def test_default_without_chunked_prefill(self):
Lianmin Zheng's avatar
Lianmin Zheng committed
92
93
94
        res = self.run_test(
            disable_radix_cache=ServerArgs.disable_radix_cache,
            disable_flashinfer=ServerArgs.disable_flashinfer,
95
            chunked_prefill_size=-1,
96
97
        )

Lianmin Zheng's avatar
Lianmin Zheng committed
98
        if os.getenv("SGLANG_IS_IN_CI", "false") == "true":
99
100
            # A100 (PCIE): 1450, H100 (SMX): 2550
            assert res["output_throughput"] > 2500
Lianmin Zheng's avatar
Lianmin Zheng committed
101

102
103
104
105
106
107
108
109
110
111
112
113
114
    def test_all_cases(self):
        for disable_radix_cache in [False, True]:
            for disable_flashinfer in [False, True]:
                for chunked_prefill_size in [-1, 2048]:
                    self.run_test(
                        disable_radix_cache=False,
                        disable_flashinfer=False,
                        chunked_prefill_size=-1,
                    )


if __name__ == "__main__":
    unittest.main()