test_serving_throughput.py 3.05 KB
Newer Older
1
import os
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
import unittest
from types import SimpleNamespace

from sglang.bench_serving import run_benchmark
from sglang.srt.utils import kill_child_process
from sglang.test.test_utils import DEFAULT_MODEL_NAME_FOR_TEST, popen_launch_server


class TestServingThroughput(unittest.TestCase):

    def run_test(self, disable_radix_cache, disable_flashinfer, chunked_prefill_size):
        # Launch the server
        other_args = []
        if disable_radix_cache:
            other_args.append("--disable-radix-cache")
        if disable_flashinfer:
            other_args.append("--disable-flashinfer")
        other_args.extend(["--chunked-prefill-size", str(chunked_prefill_size)])

        model = DEFAULT_MODEL_NAME_FOR_TEST
        base_url = "http://127.0.0.1:9157"
        process = popen_launch_server(
            model, base_url, timeout=300, other_args=other_args
        )

        # Run benchmark
        num_prompts = 400
        args = SimpleNamespace(
            backend="sglang",
            base_url=base_url,
            host=None,
            port=None,
            dataset_name="random",
            dataset_path="",
            model=None,
            tokenizer=None,
            num_prompts=num_prompts,
            sharegpt_output_len=None,
            random_input_len=4096,
            random_output_len=2048,
            random_range_ratio=0.0,
            request_rate=float("inf"),
            multi=None,
            seed=0,
            output_file=None,
            disable_tqdm=False,
            disable_stream=False,
            disable_ignore_eos=False,
            extra_request_body=None,
        )

        try:
            res = run_benchmark(args)
        finally:
            kill_child_process(process.pid)

        assert res["completed"] == num_prompts
59
        return res
60
61

    def test_default(self):
62
        res = self.run_test(
63
64
65
66
67
            disable_radix_cache=False,
            disable_flashinfer=False,
            chunked_prefill_size=-1,
        )

68
69
70
71
        if os.getenv("SGLANG_IS_IN_CI", "false") == "true":
            # A100 performance
            assert res["output_throughput"] >= 1300

72
    def test_default_without_radix_cache(self):
73
        res = self.run_test(
74
75
76
77
78
            disable_radix_cache=True,
            disable_flashinfer=False,
            chunked_prefill_size=-1,
        )

79
80
81
82
        if os.getenv("SGLANG_IS_IN_CI", "false") == "true":
            # A100 performance
            assert res["output_throughput"] >= 1400

83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
    def test_default_without_flashinfer(self):
        self.run_test(
            disable_radix_cache=False,
            disable_flashinfer=True,
            chunked_prefill_size=-1,
        )

    def test_all_cases(self):
        for disable_radix_cache in [False, True]:
            for disable_flashinfer in [False, True]:
                for chunked_prefill_size in [-1, 2048]:
                    self.run_test(
                        disable_radix_cache=False,
                        disable_flashinfer=False,
                        chunked_prefill_size=-1,
                    )


if __name__ == "__main__":
    unittest.main()