test_vllm.py 1.78 KB
Newer Older
baberabb's avatar
baberabb committed
1
from typing import List
2
3

import pytest
baberabb's avatar
baberabb committed
4

5
from lm_eval import tasks
6
7
from lm_eval.api.instance import Instance

baberabb's avatar
baberabb committed
8

9
10
11
task_manager = tasks.TaskManager()


baberabb's avatar
baberabb committed
12
@pytest.mark.skip(reason="requires CUDA")
13
class Test_VLLM:
baberabb's avatar
baberabb committed
14
    vllm = pytest.importorskip("vllm")
baberabb's avatar
baberabb committed
15
16
17
18
19
20
    try:
        from lm_eval.models.vllm_causallms import VLLM

        LM = VLLM(pretrained="EleutherAI/pythia-70m")
    except ModuleNotFoundError:
        pass
21
    # torch.use_deterministic_algorithms(True)
22
23
    task_list = task_manager.load_task_or_group(["arc_easy", "gsm8k", "wikitext"])
    multiple_choice_task = task_list["arc_easy"]  # type: ignore
baberabb's avatar
baberabb committed
24
25
    multiple_choice_task.build_all_requests(limit=10, rank=0, world_size=1)
    MULTIPLE_CH: List[Instance] = multiple_choice_task.instances
26
    generate_until_task = task_list["gsm8k"]  # type: ignore
baberabb's avatar
baberabb committed
27
    generate_until_task._config.generation_kwargs["max_gen_toks"] = 10
28
    generate_until_task.build_all_requests(limit=10, rank=0, world_size=1)
baberabb's avatar
baberabb committed
29
    generate_until: List[Instance] = generate_until_task.instances
30
    rolling_task = task_list["wikitext"]  # type: ignore
baberabb's avatar
baberabb committed
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
    rolling_task.build_all_requests(limit=10, rank=0, world_size=1)
    ROLLING: List[Instance] = rolling_task.instances

    # TODO: make proper tests
    def test_logliklihood(self) -> None:
        res = self.LM.loglikelihood(self.MULTIPLE_CH)
        assert len(res) == len(self.MULTIPLE_CH)
        for x in res:
            assert isinstance(x[0], float)

    def test_generate_until(self) -> None:
        res = self.LM.generate_until(self.generate_until)
        assert len(res) == len(self.generate_until)
        for x in res:
            assert isinstance(x, str)

    def test_logliklihood_rolling(self) -> None:
        res = self.LM.loglikelihood_rolling(self.ROLLING)
        for x in res:
            assert isinstance(x, float)