sampling_params.py 24 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3
"""Sampling parameters for text generation."""
4
import copy
5
from dataclasses import field
6
from enum import Enum, IntEnum
7
from functools import cached_property
8
from typing import Annotated, Any, Optional, Union
9

10
import msgspec
11
from pydantic.dataclasses import dataclass
Woosuk Kwon's avatar
Woosuk Kwon committed
12

13
from vllm.logger import init_logger
14
from vllm.logits_process import LogitsProcessor
15
from vllm.transformers_utils.tokenizer import AnyTokenizer
16
17
18

logger = init_logger(__name__)

19
_SAMPLING_EPS = 1e-5
20
_MAX_TEMP = 1e-2
Woosuk Kwon's avatar
Woosuk Kwon committed
21

22

23
24
25
class SamplingType(IntEnum):
    GREEDY = 0
    RANDOM = 1
Nick Hill's avatar
Nick Hill committed
26
    RANDOM_SEED = 2
27
28


29
30
# maybe make msgspec?
@dataclass
31
32
class StructuredOutputsParams:
    # One of these fields will be used to build a logit processor.
33
    json: Optional[Union[str, dict]] = None
34
    regex: Optional[str] = None
35
    choice: Optional[list[str]] = None
36
37
    grammar: Optional[str] = None
    json_object: Optional[bool] = None
38
    # These are other options that can be set.
39
40
41
    disable_fallback: bool = False
    disable_any_whitespace: bool = False
    disable_additional_properties: bool = False
42
    whitespace_pattern: Optional[str] = None
43
    structural_tag: Optional[str] = None
44

45
46
47
48
    _backend: Optional[str] = field(default=None, init=False)
    """CAUTION: Should only be set by Processor._validate_structured_output"""
    _backend_was_auto: bool = field(default=False, init=False)
    """CAUTION: Should only be set by Processor._validate_structured_output"""
49
50
51

    def __post_init__(self):
        """Validate that some fields are mutually exclusive."""
52
        count = sum([
53
54
55
            self.json is not None, self.regex is not None, self.choice
            is not None, self.grammar is not None, self.json_object is not None
        ])
56
        if count > 1:
57
            raise ValueError(
58
59
                "You can only use one kind of structured outputs constraint "
                f"but multiple are specified: {self.__dict__}")
60
61


62
63
64
65
66
class RequestOutputKind(Enum):
    # Return entire output so far in every RequestOutput
    CUMULATIVE = 0
    # Return only deltas in each RequestOutput
    DELTA = 1
67
    # Do not return intermediate RequestOutput
68
69
70
    FINAL_ONLY = 2


71
72
73
74
75
class SamplingParams(
        msgspec.Struct,
        omit_defaults=True,  # type: ignore[call-arg]
        # required for @cached_property.
        dict=True):  # type: ignore[call-arg]
76
77
78
79
80
81
    """Sampling parameters for text generation.

    Overall, we follow the sampling parameters from the OpenAI text completion
    API (https://platform.openai.com/docs/api-reference/completions/create).
    In addition, we support beam search, which is not supported by OpenAI.
    """
Woosuk Kwon's avatar
Woosuk Kwon committed
82

83
    n: int = 1
84
    """Number of output sequences to return for the given prompt."""
85
    best_of: Optional[int] = None
86
87
88
89
    """Number of output sequences that are generated from the prompt. From
    these `best_of` sequences, the top `n` sequences are returned. `best_of`
    must be greater than or equal to `n`. By default, `best_of` is set to `n`.
    Warning, this is only supported in V0."""
90
    _real_n: Optional[int] = None
91
    presence_penalty: float = 0.0
92
93
94
    """Penalizes new tokens based on whether they appear in the generated text
    so far. Values > 0 encourage the model to use new tokens, while values < 0
    encourage the model to repeat tokens."""
95
    frequency_penalty: float = 0.0
96
97
98
    """Penalizes new tokens based on their frequency in the generated text so
    far. Values > 0 encourage the model to use new tokens, while values < 0
    encourage the model to repeat tokens."""
99
    repetition_penalty: float = 1.0
100
101
102
    """Penalizes new tokens based on whether they appear in the prompt and the
    generated text so far. Values > 1 encourage the model to use new tokens,
    while values < 1 encourage the model to repeat tokens."""
103
    temperature: float = 1.0
104
105
106
    """Controls the randomness of the sampling. Lower values make the model
    more deterministic, while higher values make the model more random. Zero
    means greedy sampling."""
107
    top_p: float = 1.0
108
109
    """Controls the cumulative probability of the top tokens to consider. Must
    be in (0, 1]. Set to 1 to consider all tokens."""
110
    top_k: int = 0
111
112
    """Controls the number of top tokens to consider. Set to 0 (or -1) to
    consider all tokens."""
113
    min_p: float = 0.0
114
115
116
    """Represents the minimum probability for a token to be considered,
    relative to the probability of the most likely token. Must be in [0, 1].
    Set to 0 to disable this."""
117
    seed: Optional[int] = None
118
    """Random seed to use for the generation."""
119
    stop: Optional[Union[str, list[str]]] = None
120
121
    """String(s) that stop the generation when they are generated. The returned
    output will not contain the stop strings."""
122
    stop_token_ids: Optional[list[int]] = None
123
124
125
    """Token IDs that stop the generation when they are generated. The returned
    output will contain the stop tokens unless the stop tokens are special
    tokens."""
126
    ignore_eos: bool = False
127
128
    """Whether to ignore the EOS token and continue generating
    tokens after the EOS token is generated."""
129
    max_tokens: Optional[int] = 16
130
    """Maximum number of tokens to generate per output sequence."""
131
    min_tokens: int = 0
132
133
    """Minimum number of tokens to generate per output sequence before EOS or
    `stop_token_ids` can be generated"""
134
    logprobs: Optional[int] = None
135
136
137
138
139
140
141
    """Number of log probabilities to return per output token. When set to
    `None`, no probability is returned. If set to a non-`None` value, the
    result includes the log probabilities of the specified number of most
    likely tokens, as well as the chosen tokens. Note that the implementation
    follows the OpenAI API: The API will always return the log probability of
    the sampled token, so there may be up to `logprobs+1` elements in the
    response. When set to -1, return all `vocab_size` log probabilities."""
142
    prompt_logprobs: Optional[int] = None
143
144
    """Number of log probabilities to return per prompt token.
    When set to -1, return all `vocab_size` log probabilities."""
145
146
147
148
    # NOTE: This parameter is only exposed at the engine level for now.
    # It is not exposed in the OpenAI API server, as the OpenAI API does
    # not support returning only a list of token IDs.
    detokenize: bool = True
149
    """Whether to detokenize the output."""
150
    skip_special_tokens: bool = True
151
    """Whether to skip special tokens in the output."""
152
    spaces_between_special_tokens: bool = True
153
    """Whether to add spaces between special tokens in the output."""
154
155
    # Optional[list[LogitsProcessor]] type. We use Any here because
    # Optional[list[LogitsProcessor]] type is not supported by msgspec.
156
    logits_processors: Optional[Any] = None
157
158
    """Functions that modify logits based on previously generated tokens, and
    optionally prompt tokens as a first argument."""
159
    include_stop_str_in_output: bool = False
160
    """Whether to include the stop strings in output text."""
161
162
    truncate_prompt_tokens: Optional[Annotated[int,
                                               msgspec.Meta(ge=-1)]] = None
163
164
165
    """If set to -1, will use the truncation size supported by the model. If
    set to an integer k, will use only the last k tokens from the prompt
    (i.e., left truncation). If set to `None`, truncation is disabled."""
166
    output_kind: RequestOutputKind = RequestOutputKind.CUMULATIVE
167
168
169
170

    # The below fields are not supposed to be used as an input.
    # They are set in post_init.
    output_text_buffer_length: int = 0
171
    _all_stop_token_ids: set[int] = msgspec.field(default_factory=set)
172

173
    # Fields used to construct logits processors
174
175
    structured_outputs: Optional[StructuredOutputsParams] = None
    """Parameters for configuring structured outputs."""
176
    logit_bias: Optional[dict[int, float]] = None
177
178
    """If provided, the engine will construct a logits processor that applies
    these logit biases."""
179
    allowed_token_ids: Optional[list[int]] = None
180
181
    """If provided, the engine will construct a logits processor which only
    retains scores for the given token ids."""
182
    extra_args: Optional[dict[str, Any]] = None
183
184
185
    """Arbitrary additional args, that can be used by custom sampling
    implementations, plugins, etc. Not used by any in-tree sampling
    implementations."""
186

187
188
    # Fields used for bad words
    bad_words: Optional[list[str]] = None
189
190
191
    """Words that are not allowed to be generated. More precisely, only the
    last token of a corresponding token sequence is not allowed when the next
    generated token can complete the sequence."""
192
    _bad_words_token_ids: Optional[list[list[int]]] = None
193

194
195
196
    @staticmethod
    def from_optional(
        n: Optional[int] = 1,
197
        best_of: Optional[int] = None,
198
199
200
201
202
        presence_penalty: Optional[float] = 0.0,
        frequency_penalty: Optional[float] = 0.0,
        repetition_penalty: Optional[float] = 1.0,
        temperature: Optional[float] = 1.0,
        top_p: Optional[float] = 1.0,
203
        top_k: int = 0,
204
205
        min_p: float = 0.0,
        seed: Optional[int] = None,
206
207
208
        stop: Optional[Union[str, list[str]]] = None,
        stop_token_ids: Optional[list[int]] = None,
        bad_words: Optional[list[str]] = None,
209
210
211
212
213
214
215
216
217
        include_stop_str_in_output: bool = False,
        ignore_eos: bool = False,
        max_tokens: Optional[int] = 16,
        min_tokens: int = 0,
        logprobs: Optional[int] = None,
        prompt_logprobs: Optional[int] = None,
        detokenize: bool = True,
        skip_special_tokens: bool = True,
        spaces_between_special_tokens: bool = True,
218
        logits_processors: Optional[list[LogitsProcessor]] = None,
219
        truncate_prompt_tokens: Optional[Annotated[int,
220
221
                                                   msgspec.Meta(
                                                       ge=-1)]] = None,
222
        output_kind: RequestOutputKind = RequestOutputKind.CUMULATIVE,
223
        structured_outputs: Optional[StructuredOutputsParams] = None,
224
225
        logit_bias: Optional[Union[dict[int, float], dict[str, float]]] = None,
        allowed_token_ids: Optional[list[int]] = None,
226
        extra_args: Optional[dict[str, Any]] = None,
227
    ) -> "SamplingParams":
228
        if logit_bias is not None:
229
230
            # Convert token_id to integer
            # Clamp the bias between -100 and 100 per OpenAI API spec
231
            logit_bias = {
232
                int(token): min(100.0, max(-100.0, bias))
233
234
235
                for token, bias in logit_bias.items()
            }

236
237
        return SamplingParams(
            n=1 if n is None else n,
238
            best_of=best_of,
239
240
241
242
243
244
245
246
247
248
249
250
251
            presence_penalty=0.0
            if presence_penalty is None else presence_penalty,
            frequency_penalty=0.0
            if frequency_penalty is None else frequency_penalty,
            repetition_penalty=1.0
            if repetition_penalty is None else repetition_penalty,
            temperature=1.0 if temperature is None else temperature,
            top_p=1.0 if top_p is None else top_p,
            top_k=top_k,
            min_p=min_p,
            seed=seed,
            stop=stop,
            stop_token_ids=stop_token_ids,
252
            bad_words=bad_words,
253
254
255
256
257
258
259
260
261
262
263
            include_stop_str_in_output=include_stop_str_in_output,
            ignore_eos=ignore_eos,
            max_tokens=max_tokens,
            min_tokens=min_tokens,
            logprobs=logprobs,
            prompt_logprobs=prompt_logprobs,
            detokenize=detokenize,
            skip_special_tokens=skip_special_tokens,
            spaces_between_special_tokens=spaces_between_special_tokens,
            logits_processors=logits_processors,
            truncate_prompt_tokens=truncate_prompt_tokens,
264
            output_kind=output_kind,
265
            structured_outputs=structured_outputs,
266
267
            logit_bias=logit_bias,
            allowed_token_ids=allowed_token_ids,
268
            extra_args=extra_args,
269
270
        )

271
    def __post_init__(self) -> None:
272
273
274
275
276
277
278
279
280
281
282
283
284
285
        # how we deal with `best_of``:
        # if `best_of`` is not set, we default to `n`;
        # if `best_of`` is set, we set `n`` to `best_of`,
        # and set `_real_n`` to the original `n`.
        # when we return the result, we will check
        # if we need to return `n` or `_real_n` results
        if self.best_of:
            if self.best_of < self.n:
                raise ValueError(
                    f"best_of must be greater than or equal to n, "
                    f"got n={self.n} and best_of={self.best_of}.")
            if not self._real_n:
                self._real_n = self.n
                self.n = self.best_of
286

287
        if 0 < self.temperature < _MAX_TEMP:
288
289
290
            logger.warning(
                "temperature %s is less than %s, which may cause numerical "
                "errors nan or inf in tensors. We have maxed it out to %s.",
291
292
                self.temperature, _MAX_TEMP, _MAX_TEMP)
            self.temperature = max(self.temperature, _MAX_TEMP)
293

294
        if self.seed == -1:
295
            self.seed = None
296

297
        if self.stop is None:
298
            self.stop = []
299
300
        elif isinstance(self.stop, str):
            self.stop = [self.stop]
301

302
        if self.stop_token_ids is None:
303
            self.stop_token_ids = []
304
305
306
307

        if self.bad_words is None:
            self.bad_words = []

308
309
310
311
312
        if self.logprobs is True:
            self.logprobs = 1

        if self.prompt_logprobs is True:
            self.prompt_logprobs = 1
313

314
315
        # Number of characters to hold back for stop string evaluation
        # until sequence is finished.
316
        if self.stop and not self.include_stop_str_in_output:
317
318
            self.output_text_buffer_length = max(len(s) for s in self.stop) - 1

319
        self._verify_args()
320
321
322
323

        if self.temperature < _SAMPLING_EPS:
            # Zero temperature means greedy sampling.
            self.top_p = 1.0
324
            self.top_k = 0
325
326
            self.min_p = 0.0
            self._verify_greedy_sampling()
327

328
        # eos_token_id is added to this by the engine
329
        self._all_stop_token_ids.update(self.stop_token_ids)
330
331

    def _verify_args(self) -> None:
332
333
334
        if not isinstance(self.n, int):
            raise ValueError(f"n must be an int, but is of "
                             f"type {type(self.n)}")
335
336
        if self.n < 1:
            raise ValueError(f"n must be at least 1, got {self.n}.")
337
338
339
340
341
342
343
344
345
346
347
        if self.best_of is not None:
            if not isinstance(self.best_of, int):
                raise ValueError(
                    f"best_of must be an integer, got {type(self.best_of)}")
            if self.best_of < 1:
                raise ValueError(
                    f"best_of must be at least 1, got {self.best_of}")
            if self.best_of < self.n:
                raise ValueError(
                    f"best_of must be greater than or equal to n, "
                    f"got n={self.n} and best_of={self.best_of}.")
348
349
350
351
352
353
        if not -2.0 <= self.presence_penalty <= 2.0:
            raise ValueError("presence_penalty must be in [-2, 2], got "
                             f"{self.presence_penalty}.")
        if not -2.0 <= self.frequency_penalty <= 2.0:
            raise ValueError("frequency_penalty must be in [-2, 2], got "
                             f"{self.frequency_penalty}.")
354
355
356
357
        if self.repetition_penalty <= 0.0:
            raise ValueError(
                "repetition_penalty must be greater than zero, got "
                f"{self.repetition_penalty}.")
358
359
360
361
362
        if self.temperature < 0.0:
            raise ValueError(
                f"temperature must be non-negative, got {self.temperature}.")
        if not 0.0 < self.top_p <= 1.0:
            raise ValueError(f"top_p must be in (0, 1], got {self.top_p}.")
363
364
365
        # quietly accept -1 as disabled, but prefer 0
        if self.top_k < -1:
            raise ValueError(f"top_k must be 0 (disable), or at least 1, "
366
                             f"got {self.top_k}.")
367
368
369
        if not isinstance(self.top_k, int):
            raise TypeError(
                f"top_k must be an integer, got {type(self.top_k).__name__}")
Roy's avatar
Roy committed
370
371
372
        if not 0.0 <= self.min_p <= 1.0:
            raise ValueError("min_p must be in [0, 1], got "
                             f"{self.min_p}.")
373
        if self.max_tokens is not None and self.max_tokens < 1:
374
375
            raise ValueError(
                f"max_tokens must be at least 1, got {self.max_tokens}.")
376
377
378
379
380
381
382
        if self.min_tokens < 0:
            raise ValueError(f"min_tokens must be greater than or equal to 0, "
                             f"got {self.min_tokens}.")
        if self.max_tokens is not None and self.min_tokens > self.max_tokens:
            raise ValueError(
                f"min_tokens must be less than or equal to "
                f"max_tokens={self.max_tokens}, got {self.min_tokens}.")
383
384
        if (self.logprobs is not None and self.logprobs != -1
                and self.logprobs < 0):
385
            raise ValueError(
386
                f"logprobs must be non-negative or -1, got {self.logprobs}.")
387
388
389
390
391
        if (self.prompt_logprobs is not None and self.prompt_logprobs != -1
                and self.prompt_logprobs < 0):
            raise ValueError(
                f"prompt_logprobs must be non-negative or -1, got "
                f"{self.prompt_logprobs}.")
392
        if (self.truncate_prompt_tokens is not None
393
394
395
396
397
                and (self.truncate_prompt_tokens == 0
                     or self.truncate_prompt_tokens < -1)):
            raise ValueError(
                f"truncate_prompt_tokens must be an integer >= 1 or -1, "
                f"got {self.truncate_prompt_tokens}")
398
399
400
401
        assert isinstance(self.stop_token_ids, list)
        if not all(isinstance(st_id, int) for st_id in self.stop_token_ids):
            raise ValueError(f"stop_token_ids must contain only integers, "
                             f"got {self.stop_token_ids}.")
402
        assert isinstance(self.stop, list)
403
404
        if any(not stop_str for stop_str in self.stop):
            raise ValueError("stop cannot contain an empty string.")
405
406
407
408
        if self.stop and not self.detokenize:
            raise ValueError(
                "stop strings are only supported when detokenize is True. "
                "Set detokenize=True to use stop.")
409
410
411
        if self.best_of != self._real_n and self.output_kind == (
                RequestOutputKind.DELTA):
            raise ValueError("best_of must equal n to use output_kind=DELTA")
412
413

    def _verify_greedy_sampling(self) -> None:
414
415
416
        if self.n > 1:
            raise ValueError("n must be 1 when using greedy sampling, "
                             f"got {self.n}.")
417

418
    def update_from_generation_config(
419
            self,
420
            generation_config: dict[str, Any],
421
            model_eos_token_id: Optional[int] = None) -> None:
422
        """Update if there are non-default values from generation_config"""
423
424
425
426

        if model_eos_token_id is not None:
            # Add the eos token id into the sampling_params to support
            # min_tokens processing.
427
            self._all_stop_token_ids.add(model_eos_token_id)
428

429
        # Update eos_token_id for generation
430
        if (eos_ids := generation_config.get("eos_token_id")) is not None:
431
            # it can be either int or list of int
432
433
434
435
436
437
438
            eos_ids = {eos_ids} if isinstance(eos_ids, int) else set(eos_ids)
            if model_eos_token_id is not None:
                # We don't need to include the primary eos_token_id in
                # stop_token_ids since it's handled separately for stopping
                # purposes.
                eos_ids.discard(model_eos_token_id)
            if eos_ids:
439
                self._all_stop_token_ids.update(eos_ids)
440
441
442
                if not self.ignore_eos:
                    eos_ids.update(self.stop_token_ids)
                    self.stop_token_ids = list(eos_ids)
443

444
    def update_from_tokenizer(self, tokenizer: AnyTokenizer) -> None:
445
        if not self.bad_words:
446
            return
447
        self._bad_words_token_ids = []
448
449
450
451
452
453
454
        for bad_word in self.bad_words:
            # To prohibit words both at the beginning
            # and in the middle of text
            # (related to add_prefix_space tokenizer parameter)
            for add_prefix_space in [False, True]:
                prefix = " " if add_prefix_space else ""
                prompt = prefix + bad_word.lstrip()
455
456
                prompt_token_ids = tokenizer.encode(text=prompt,
                                                    add_special_tokens=False)
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479

                # If no space at the beginning
                # or if prefix space produces a new word token
                if (not add_prefix_space) or (
                        add_prefix_space and prompt_token_ids[0]
                        != self._bad_words_token_ids[-1][0]
                        and len(prompt_token_ids) == len(
                            self._bad_words_token_ids[-1])):
                    self._bad_words_token_ids.append(prompt_token_ids)

        invalid_token_ids = [
            token_id for bad_words_token_ids in self._bad_words_token_ids
            for token_id in bad_words_token_ids
            if token_id < 0 or token_id > tokenizer.max_token_id
        ]
        if len(invalid_token_ids) > 0:
            raise ValueError(
                f"The model vocabulary size is {tokenizer.max_token_id+1},"
                f" but the following tokens"
                f" were specified as bad: {invalid_token_ids}."
                f" All token id values should be integers satisfying:"
                f" 0 <= token_id <= {tokenizer.max_token_id}.")

480
481
482
483
    @cached_property
    def sampling_type(self) -> SamplingType:
        if self.temperature < _SAMPLING_EPS:
            return SamplingType.GREEDY
Nick Hill's avatar
Nick Hill committed
484
485
        if self.seed is not None:
            return SamplingType.RANDOM_SEED
486
487
        return SamplingType.RANDOM

488
    @property
489
    def all_stop_token_ids(self) -> set[int]:
490
491
        return self._all_stop_token_ids

492
    @property
493
    def bad_words_token_ids(self) -> Optional[list[list[int]]]:
494
495
496
        # For internal use only. Backward compatibility not guaranteed
        return self._bad_words_token_ids

497
    def clone(self) -> "SamplingParams":
498
        """Deep copy, but maybe not the LogitsProcessor objects.
499

500
501
502
        LogitsProcessor objects may contain an arbitrary, nontrivial amount of
        data that is expensive to copy. However, if not copied, the processor
        needs to support parallel decoding for multiple sequences
503
504
505
506
        See https://github.com/vllm-project/vllm/issues/3087
        """

        logit_processor_refs = None if self.logits_processors is None else {
507
            id(lp): lp.clone() if hasattr(lp, 'clone') else lp
508
509
510
511
            for lp in self.logits_processors
        }
        return copy.deepcopy(self, memo=logit_processor_refs)

512
    def __repr__(self) -> str:
513
514
515
516
517
518
519
520
521
        return (
            f"SamplingParams(n={self.n}, "
            f"presence_penalty={self.presence_penalty}, "
            f"frequency_penalty={self.frequency_penalty}, "
            f"repetition_penalty={self.repetition_penalty}, "
            f"temperature={self.temperature}, "
            f"top_p={self.top_p}, "
            f"top_k={self.top_k}, "
            f"min_p={self.min_p}, "
Nick Hill's avatar
Nick Hill committed
522
            f"seed={self.seed}, "
523
524
            f"stop={self.stop}, "
            f"stop_token_ids={self.stop_token_ids}, "
525
            f"bad_words={self.bad_words}, "
526
527
528
            f"include_stop_str_in_output={self.include_stop_str_in_output}, "
            f"ignore_eos={self.ignore_eos}, "
            f"max_tokens={self.max_tokens}, "
529
            f"min_tokens={self.min_tokens}, "
530
531
532
533
            f"logprobs={self.logprobs}, "
            f"prompt_logprobs={self.prompt_logprobs}, "
            f"skip_special_tokens={self.skip_special_tokens}, "
            "spaces_between_special_tokens="
534
            f"{self.spaces_between_special_tokens}, "
535
            f"truncate_prompt_tokens={self.truncate_prompt_tokens}, "
536
            f"structured_outputs={self.structured_outputs}, "
537
            f"extra_args={self.extra_args})")
538
539
540
541
542
543
544
545
546
547
548
549


class BeamSearchParams(
        msgspec.Struct,
        omit_defaults=True,  # type: ignore[call-arg]
        # required for @cached_property.
        dict=True):  # type: ignore[call-arg]
    """Beam search parameters for text generation."""
    beam_width: int
    max_tokens: int
    ignore_eos: bool = False
    temperature: float = 0.0
550
    length_penalty: float = 1.0
551
    include_stop_str_in_output: bool = False