"vllm/model_executor/models/h2ovl.py" did not exist on "1f1b6d6eda3ea5fbdf4566632ac8a9fa61b31593"
nemotron.py 9.54 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3

4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
# Copyright 2024 HuggingFace Inc. team. All rights reserved.
# Copyright (c) 2024, NVIDIA CORPORATION. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
"""Nemotron model configuration"""

from transformers import PretrainedConfig
from transformers.utils import logging

logger = logging.get_logger(__name__)


class NemotronConfig(PretrainedConfig):
    r"""
    This is the configuration class to store the configuration of a
29
    [`NemotronModel`]. It is used to instantiate a Nemotron model
30
31
32
33
34
35
36
37
38
39
    according to the specified arguments, defining the model architecture.
    Instantiating a configuration with the defaults will yield a similar
    configuration to that of the Nemotron-8B.

    Configuration objects inherit from [`PretrainedConfig`] and can be
    used to control the model outputs. Read the documentation from
    [`PretrainedConfig`] for more information.


    Args:
40
        vocab_size (`int`, *optional*, defaults to 256000):
41
42
43
            Vocabulary size of the Nemotron model. Defines the number of
            different tokens that can be represented by the
            `inputs_ids` passed when calling [`NemotronModel`]
44
        hidden_size (`int`, *optional*, defaults to 6144):
45
            Dimension of the hidden representations.
46
        intermediate_size (`int`, *optional*, defaults to 24576):
47
48
49
            Dimension of the MLP representations.
        num_hidden_layers (`int`, *optional*, defaults to 32):
            Number of hidden layers in the Transformer decoder.
50
        num_attention_heads (`int`, *optional*, defaults to 48):
51
52
            Number of attention heads for each attention layer in the
            Transformer decoder.
53
        head_dim (`int`, *optional*):
54
55
56
57
58
59
60
61
62
63
64
            Projection weights dimension in multi-head attention. Set to
            hidden_size // num_attention_heads if None
        num_key_value_heads (`int`, *optional*):
            This is the number of key_value heads that should be used to
            implement Grouped Query Attention. If
            `num_key_value_heads=num_attention_heads`, the model will use
            Multi Head Attention (MHA), if
            `num_key_value_heads=1 the model will use Multi Query Attention
            (MQA) otherwise GQA is used. When converting a multi-head
            checkpoint to a GQA checkpoint, each group key and value
            head should be constructed by meanpooling all the original
65
            heads within that group. For more details checkout
66
67
            [this paper](https://arxiv.org/pdf/2305.13245.pdf). If it
            is not specified, will default to `num_attention_heads`.
68
        hidden_act (`str` or `function`, *optional*, defaults to `"relu2"`):
69
70
            The non-linear activation function (function or string) in the
            decoder.
71
        max_position_embeddings (`int`, *optional*, defaults to 4096):
72
73
            The maximum sequence length that this model might ever be used
            with.
74
        initializer_range (`float`, *optional*, defaults to 0.0134):
75
76
            The standard deviation of the truncated_normal_initializer for
            initializing all weight matrices.
77
        norm_eps (`float`, *optional*, defaults to 1e-05):
78
79
80
81
82
83
84
            The epsilon used by the normalization layers.
        use_cache (`bool`, *optional*, defaults to `True`):
            Whether or not the model should return the last key/values
            attentions (not used by all models). Only relevant if
            `config.is_decoder=True`.
        pad_token_id (`int`, *optional*):
            Padding token id.
85
        bos_token_id (`int`, *optional*, defaults to 2):
86
            Beginning of stream token id.
87
        eos_token_id (`int`, *optional*, defaults to 3):
88
89
90
            End of stream token id.
        tie_word_embeddings (`bool`, *optional*, defaults to `False`):
            Whether to tie weight embeddings
91
        rope_parameters (`dict`, *optional*):
92
93
94
95
96
97
98
99
            The parameters of the RoPE embeddings. Expected contents:
                `rope_theta` (`float`): The base period of the RoPE embeddings.
                `rope_type` (`str`):
                    The sub-variant of RoPE to use. Can be one of ['default', 'linear',
                    'dynamic', 'yarn', 'longrope', 'llama3'], with 'default' being the
                    original RoPE implementation.
                `partial_rotary_factor` (`float`, *optional*, defaults to 0.5):
                    Percentage of the query and keys which will have rotary embedding.
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
        attention_bias (`bool`, *optional*, defaults to `False`):
            Whether to use a bias in the query, key, value and output
            projection layers during self-attention.
        attention_dropout (`float`, *optional*, defaults to 0.0):
            The dropout ratio for the attention probabilities.
        mlp_bias (`bool`, *optional*, defaults to `False`):
            Whether to use a bias in up_proj and down_proj layers in the MLP
            layers.

    ```python
    >>> from transformers import NemotronModel, NemotronConfig
    >>> # Initializing a Nemotron nemotron-15b style configuration
    >>> configuration = NemotronConfig()
    >>> # Initializing a model from the nemotron-15b style configuration
    >>> model = NemotronModel(configuration)
    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```"""

    model_type = "nemotron"
    keys_to_ignore_at_inference = ["past_key_values"]

    def __init__(
        self,
        vocab_size=256000,
        hidden_size=6144,
        intermediate_size=24576,
        num_hidden_layers=32,
        num_attention_heads=48,
        head_dim=None,
        num_key_value_heads=None,
        hidden_act="relu2",
        max_position_embeddings=4096,
        initializer_range=0.0134,
        norm_eps=1e-5,
        use_cache=True,
        pad_token_id=None,
        bos_token_id=2,
        eos_token_id=3,
        tie_word_embeddings=False,
140
        rope_parameters=None,
141
142
143
144
145
146
147
148
149
150
151
        attention_bias=False,
        attention_dropout=0.0,
        mlp_bias=False,
        **kwargs,
    ):
        self.vocab_size = vocab_size
        self.max_position_embeddings = max_position_embeddings
        self.hidden_size = hidden_size
        self.intermediate_size = intermediate_size
        self.num_hidden_layers = num_hidden_layers
        self.num_attention_heads = num_attention_heads
152
        head_dim = head_dim or kwargs.get("kv_channels")
153
154
155
        self.head_dim = (
            head_dim if head_dim is not None else (hidden_size // num_attention_heads)
        )
156
157
158
159
160
161
162
163
164
165

        # for backward compatibility
        if num_key_value_heads is None:
            num_key_value_heads = num_attention_heads

        self.num_key_value_heads = num_key_value_heads
        self.hidden_act = hidden_act
        self.initializer_range = initializer_range
        self.norm_eps = norm_eps
        self.use_cache = use_cache
166
167
168
169
170
171
        # Try to set `rope_scaling` if available, otherwise use `rope_parameters`
        rope_scaling = kwargs.pop("rope_scaling", None)
        rope_parameters = rope_scaling or rope_parameters or {"rope_type": "default"}
        rope_theta = kwargs.pop("rope_theta", 10000.0)
        if "rope_theta" not in rope_parameters:
            rope_parameters["rope_theta"] = rope_theta
172
        # for backward compatibility
173
174
175
        partial_rotary_factor = (
            kwargs.get("rope_percent")
            or kwargs.get("rope_percentage")
176
177
            or kwargs.get("partial_rotary_factor")
            or 0.5
178
        )
179
180
181
        if "partial_rotary_factor" not in rope_parameters:
            rope_parameters["partial_rotary_factor"] = partial_rotary_factor
        self.rope_parameters = rope_parameters
182
        self._rope_parameters_validation()
183
184
185
186
187
188
189
190
191
192
193
194
        self.attention_bias = attention_bias
        self.attention_dropout = attention_dropout
        self.mlp_bias = mlp_bias

        super().__init__(
            pad_token_id=pad_token_id,
            bos_token_id=bos_token_id,
            eos_token_id=eos_token_id,
            tie_word_embeddings=tie_word_embeddings,
            **kwargs,
        )

195
    def _rope_parameters_validation(self):
196
        """
197
        Validate the `rope_parameters` configuration.
198
        """
199
        if self.rope_parameters is None:
200
201
            return

202
203
204
205
        rope_type: str | None = self.rope_parameters.get("rope_type", None)
        factor: float | None = self.rope_parameters.get("factor", None)

        if rope_type not in {"default", "linear", "dynamic"}:
206
            raise ValueError(
207
208
                "`rope_type` must be one of ['default', 'linear', 'dynamic'], "
                f"got {rope_type}"
209
            )
210
211
212
213
214
215
216
217
218
219
220
        if rope_type != "default":
            if factor is None:
                raise ValueError(
                    "If `rope_type` is not 'default', `rope_parameters` "
                    "must include a `factor` field. Got `None`."
                )
            if not isinstance(factor, float) or factor <= 1.0:
                raise ValueError(
                    "`rope_parameters`'s factor field must be a float > 1, got "
                    f"{factor}"
                )