test_tokenizer.py 959 Bytes
Newer Older
1
2
# SPDX-License-Identifier: Apache-2.0

3
import pytest
4
import os
5
6
7
from transformers import PreTrainedTokenizerBase

from vllm.transformers_utils.tokenizer import get_tokenizer
8
from ..utils import models_path_prefix
9

10
11
12
13
14
15
# TOKENIZER_NAMES = [
#     os.path.join(models_path_prefix, "facebook/opt-125m"),
#     os.path.join(models_path_prefix, "gpt2"),
# ]

# export HF_ENDPOINT=https://hf-mirror.com
16
TOKENIZER_NAMES = [
17
18
    "facebook/opt-125m",
    "gpt2",
19
20
21
22
23
24
]


@pytest.mark.parametrize("tokenizer_name", TOKENIZER_NAMES)
def test_tokenizer_revision(tokenizer_name: str):
    # Assume that "main" branch always exists
zhuwenwen's avatar
zhuwenwen committed
25
26
    # tokenizer = get_tokenizer(tokenizer_name, revision="main")
    tokenizer = get_tokenizer(tokenizer_name)
27
28
29
30
31
    assert isinstance(tokenizer, PreTrainedTokenizerBase)

    # Assume that "never" branch always does not exist
    with pytest.raises(OSError, match='not a valid git identifier'):
        get_tokenizer(tokenizer_name, revision="never")