test_torch_compile.py 1.88 KB
Newer Older
1
2
3
import unittest
from types import SimpleNamespace

4
5
import requests

6
7
from sglang.srt.utils import kill_child_process
from sglang.test.run_eval import run_eval
8
9
from sglang.test.test_utils import (
    DEFAULT_MODEL_NAME_FOR_TEST,
10
11
    DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
    DEFAULT_URL_FOR_TEST,
12
13
    popen_launch_server,
)
14
15


16
class TestTorchCompile(unittest.TestCase):
17
18
    @classmethod
    def setUpClass(cls):
Ying Sheng's avatar
Ying Sheng committed
19
        cls.model = DEFAULT_MODEL_NAME_FOR_TEST
20
        cls.base_url = DEFAULT_URL_FOR_TEST
21
        cls.process = popen_launch_server(
22
23
24
            cls.model,
            cls.base_url,
            timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
25
            other_args=["--enable-torch-compile", "--disable-radix-cache"],
26
27
28
29
30
31
32
33
34
35
36
        )

    @classmethod
    def tearDownClass(cls):
        kill_child_process(cls.process.pid)

    def test_mmlu(self):
        args = SimpleNamespace(
            base_url=self.base_url,
            model=self.model,
            eval_name="mmlu",
37
38
            num_examples=32,
            num_threads=32,
39
40
41
        )

        metrics = run_eval(args)
42
        assert metrics["score"] >= 0.6
43

44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
    def run_decode(self, max_new_tokens):
        response = requests.post(
            self.base_url + "/generate",
            json={
                "text": "The capital of France is",
                "sampling_params": {
                    "temperature": 0,
                    "max_new_tokens": max_new_tokens,
                },
                "ignore_eos": True,
            },
        )
        return response.json()

    def test_throughput(self):
        import time

        max_tokens = 256

        tic = time.time()
        res = self.run_decode(max_tokens)
        tok = time.time()
        print(res["text"])
        throughput = max_tokens / (tok - tic)
        print(f"Throughput: {throughput} tokens/s")
        assert throughput >= 152

71
72

if __name__ == "__main__":
Lianmin Zheng's avatar
Lianmin Zheng committed
73
    unittest.main()