sampling_params.py 24.9 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3
"""Sampling parameters for text generation."""
4
import copy
5
from dataclasses import dataclass
6
from enum import Enum, IntEnum
7
from functools import cached_property
8
from typing import Annotated, Any, Optional, Union
9

10
import msgspec
11
from pydantic import BaseModel
Woosuk Kwon's avatar
Woosuk Kwon committed
12

13
from vllm.logger import init_logger
14
from vllm.logits_process import LogitsProcessor
15
from vllm.transformers_utils.tokenizer import AnyTokenizer
16
17
18

logger = init_logger(__name__)

19
_SAMPLING_EPS = 1e-5
20
_MAX_TEMP = 1e-2
Woosuk Kwon's avatar
Woosuk Kwon committed
21

22

23
24
25
class SamplingType(IntEnum):
    GREEDY = 0
    RANDOM = 1
Nick Hill's avatar
Nick Hill committed
26
    RANDOM_SEED = 2
27
28


29
30
31
32
# maybe make msgspec?
@dataclass
class GuidedDecodingParams:
    """One of these fields will be used to build a logit processor."""
33
    json: Optional[Union[str, dict]] = None
34
    regex: Optional[str] = None
35
    choice: Optional[list[str]] = None
36
37
38
39
    grammar: Optional[str] = None
    json_object: Optional[bool] = None
    """These are other options that can be set"""
    backend: Optional[str] = None
40
41
42
43
    backend_was_auto: bool = False
    disable_fallback: bool = False
    disable_any_whitespace: bool = False
    disable_additional_properties: bool = False
44
    whitespace_pattern: Optional[str] = None
45
    structural_tag: Optional[str] = None
46
47
48

    @staticmethod
    def from_optional(
49
        json: Optional[Union[dict, BaseModel, str]] = None,
50
        regex: Optional[str] = None,
51
        choice: Optional[list[str]] = None,
52
53
54
55
        grammar: Optional[str] = None,
        json_object: Optional[bool] = None,
        backend: Optional[str] = None,
        whitespace_pattern: Optional[str] = None,
56
        structural_tag: Optional[str] = None,
57
    ) -> Optional["GuidedDecodingParams"]:
58
59
        if all(arg is None for arg in (json, regex, choice, grammar,
                                       json_object, structural_tag)):
60
            return None
61
62
63
64
65
66
67
68
69
70
71
        # Extract json schemas from pydantic models
        if isinstance(json, (BaseModel, type(BaseModel))):
            json = json.model_json_schema()
        return GuidedDecodingParams(
            json=json,
            regex=regex,
            choice=choice,
            grammar=grammar,
            json_object=json_object,
            backend=backend,
            whitespace_pattern=whitespace_pattern,
72
            structural_tag=structural_tag,
73
74
75
76
77
78
79
80
81
82
83
84
85
86
        )

    def __post_init__(self):
        """Validate that some fields are mutually exclusive."""
        guide_count = sum([
            self.json is not None, self.regex is not None, self.choice
            is not None, self.grammar is not None, self.json_object is not None
        ])
        if guide_count > 1:
            raise ValueError(
                "You can only use one kind of guided decoding but multiple are "
                f"specified: {self.__dict__}")


87
88
89
90
91
class RequestOutputKind(Enum):
    # Return entire output so far in every RequestOutput
    CUMULATIVE = 0
    # Return only deltas in each RequestOutput
    DELTA = 1
92
    # Do not return intermediate RequestOutput
93
94
95
    FINAL_ONLY = 2


96
97
98
99
100
class SamplingParams(
        msgspec.Struct,
        omit_defaults=True,  # type: ignore[call-arg]
        # required for @cached_property.
        dict=True):  # type: ignore[call-arg]
101
102
103
104
105
106
    """Sampling parameters for text generation.

    Overall, we follow the sampling parameters from the OpenAI text completion
    API (https://platform.openai.com/docs/api-reference/completions/create).
    In addition, we support beam search, which is not supported by OpenAI.
    """
Woosuk Kwon's avatar
Woosuk Kwon committed
107

108
    n: int = 1
109
    """Number of output sequences to return for the given prompt."""
110
    best_of: Optional[int] = None
111
112
113
114
    """Number of output sequences that are generated from the prompt. From
    these `best_of` sequences, the top `n` sequences are returned. `best_of`
    must be greater than or equal to `n`. By default, `best_of` is set to `n`.
    Warning, this is only supported in V0."""
115
    _real_n: Optional[int] = None
116
    presence_penalty: float = 0.0
117
118
119
    """Penalizes new tokens based on whether they appear in the generated text
    so far. Values > 0 encourage the model to use new tokens, while values < 0
    encourage the model to repeat tokens."""
120
    frequency_penalty: float = 0.0
121
122
123
    """Penalizes new tokens based on their frequency in the generated text so
    far. Values > 0 encourage the model to use new tokens, while values < 0
    encourage the model to repeat tokens."""
124
    repetition_penalty: float = 1.0
125
126
127
    """Penalizes new tokens based on whether they appear in the prompt and the
    generated text so far. Values > 1 encourage the model to use new tokens,
    while values < 1 encourage the model to repeat tokens."""
128
    temperature: float = 1.0
129
130
131
    """Controls the randomness of the sampling. Lower values make the model
    more deterministic, while higher values make the model more random. Zero
    means greedy sampling."""
132
    top_p: float = 1.0
133
134
    """Controls the cumulative probability of the top tokens to consider. Must
    be in (0, 1]. Set to 1 to consider all tokens."""
135
    top_k: int = 0
136
137
    """Controls the number of top tokens to consider. Set to 0 (or -1) to
    consider all tokens."""
138
    min_p: float = 0.0
139
140
141
    """Represents the minimum probability for a token to be considered,
    relative to the probability of the most likely token. Must be in [0, 1].
    Set to 0 to disable this."""
142
    seed: Optional[int] = None
143
    """Random seed to use for the generation."""
144
    stop: Optional[Union[str, list[str]]] = None
145
146
    """String(s) that stop the generation when they are generated. The returned
    output will not contain the stop strings."""
147
    stop_token_ids: Optional[list[int]] = None
148
149
150
    """Token IDs that stop the generation when they are generated. The returned
    output will contain the stop tokens unless the stop tokens are special
    tokens."""
151
    ignore_eos: bool = False
152
153
    """Whether to ignore the EOS token and continue generating
    tokens after the EOS token is generated."""
154
    max_tokens: Optional[int] = 16
155
    """Maximum number of tokens to generate per output sequence."""
156
    min_tokens: int = 0
157
158
    """Minimum number of tokens to generate per output sequence before EOS or
    `stop_token_ids` can be generated"""
159
    logprobs: Optional[int] = None
160
161
162
163
164
165
166
    """Number of log probabilities to return per output token. When set to
    `None`, no probability is returned. If set to a non-`None` value, the
    result includes the log probabilities of the specified number of most
    likely tokens, as well as the chosen tokens. Note that the implementation
    follows the OpenAI API: The API will always return the log probability of
    the sampled token, so there may be up to `logprobs+1` elements in the
    response. When set to -1, return all `vocab_size` log probabilities."""
167
    prompt_logprobs: Optional[int] = None
168
169
    """Number of log probabilities to return per prompt token.
    When set to -1, return all `vocab_size` log probabilities."""
170
171
172
173
    # NOTE: This parameter is only exposed at the engine level for now.
    # It is not exposed in the OpenAI API server, as the OpenAI API does
    # not support returning only a list of token IDs.
    detokenize: bool = True
174
    """Whether to detokenize the output."""
175
    skip_special_tokens: bool = True
176
    """Whether to skip special tokens in the output."""
177
    spaces_between_special_tokens: bool = True
178
    """Whether to add spaces between special tokens in the output."""
179
180
    # Optional[list[LogitsProcessor]] type. We use Any here because
    # Optional[list[LogitsProcessor]] type is not supported by msgspec.
181
    logits_processors: Optional[Any] = None
182
183
    """Functions that modify logits based on previously generated tokens, and
    optionally prompt tokens as a first argument."""
184
    include_stop_str_in_output: bool = False
185
    """Whether to include the stop strings in output text."""
186
187
    truncate_prompt_tokens: Optional[Annotated[int,
                                               msgspec.Meta(ge=-1)]] = None
188
189
190
    """If set to -1, will use the truncation size supported by the model. If
    set to an integer k, will use only the last k tokens from the prompt
    (i.e., left truncation). If set to `None`, truncation is disabled."""
191
    output_kind: RequestOutputKind = RequestOutputKind.CUMULATIVE
192
193
194
195

    # The below fields are not supposed to be used as an input.
    # They are set in post_init.
    output_text_buffer_length: int = 0
196
    _all_stop_token_ids: set[int] = msgspec.field(default_factory=set)
197

198
199
    # Fields used to construct logits processors
    guided_decoding: Optional[GuidedDecodingParams] = None
200
201
    """If provided, the engine will construct a guided decoding logits
    processor from these parameters."""
202
    logit_bias: Optional[dict[int, float]] = None
203
204
    """If provided, the engine will construct a logits processor that applies
    these logit biases."""
205
    allowed_token_ids: Optional[list[int]] = None
206
207
    """If provided, the engine will construct a logits processor which only
    retains scores for the given token ids."""
208
    extra_args: Optional[dict[str, Any]] = None
209
210
211
    """Arbitrary additional args, that can be used by custom sampling
    implementations, plugins, etc. Not used by any in-tree sampling
    implementations."""
212

213
214
    # Fields used for bad words
    bad_words: Optional[list[str]] = None
215
216
217
    """Words that are not allowed to be generated. More precisely, only the
    last token of a corresponding token sequence is not allowed when the next
    generated token can complete the sequence."""
218
    _bad_words_token_ids: Optional[list[list[int]]] = None
219

220
221
222
    @staticmethod
    def from_optional(
        n: Optional[int] = 1,
223
        best_of: Optional[int] = None,
224
225
226
227
228
        presence_penalty: Optional[float] = 0.0,
        frequency_penalty: Optional[float] = 0.0,
        repetition_penalty: Optional[float] = 1.0,
        temperature: Optional[float] = 1.0,
        top_p: Optional[float] = 1.0,
229
        top_k: int = 0,
230
231
        min_p: float = 0.0,
        seed: Optional[int] = None,
232
233
234
        stop: Optional[Union[str, list[str]]] = None,
        stop_token_ids: Optional[list[int]] = None,
        bad_words: Optional[list[str]] = None,
235
236
237
238
239
240
241
242
243
        include_stop_str_in_output: bool = False,
        ignore_eos: bool = False,
        max_tokens: Optional[int] = 16,
        min_tokens: int = 0,
        logprobs: Optional[int] = None,
        prompt_logprobs: Optional[int] = None,
        detokenize: bool = True,
        skip_special_tokens: bool = True,
        spaces_between_special_tokens: bool = True,
244
        logits_processors: Optional[list[LogitsProcessor]] = None,
245
        truncate_prompt_tokens: Optional[Annotated[int,
246
247
                                                   msgspec.Meta(
                                                       ge=-1)]] = None,
248
        output_kind: RequestOutputKind = RequestOutputKind.CUMULATIVE,
249
        guided_decoding: Optional[GuidedDecodingParams] = None,
250
251
        logit_bias: Optional[Union[dict[int, float], dict[str, float]]] = None,
        allowed_token_ids: Optional[list[int]] = None,
252
        extra_args: Optional[dict[str, Any]] = None,
253
    ) -> "SamplingParams":
254
        if logit_bias is not None:
255
256
            # Convert token_id to integer
            # Clamp the bias between -100 and 100 per OpenAI API spec
257
            logit_bias = {
258
                int(token): min(100.0, max(-100.0, bias))
259
260
261
                for token, bias in logit_bias.items()
            }

262
263
        return SamplingParams(
            n=1 if n is None else n,
264
            best_of=best_of,
265
266
267
268
269
270
271
272
273
274
275
276
277
            presence_penalty=0.0
            if presence_penalty is None else presence_penalty,
            frequency_penalty=0.0
            if frequency_penalty is None else frequency_penalty,
            repetition_penalty=1.0
            if repetition_penalty is None else repetition_penalty,
            temperature=1.0 if temperature is None else temperature,
            top_p=1.0 if top_p is None else top_p,
            top_k=top_k,
            min_p=min_p,
            seed=seed,
            stop=stop,
            stop_token_ids=stop_token_ids,
278
            bad_words=bad_words,
279
280
281
282
283
284
285
286
287
288
289
            include_stop_str_in_output=include_stop_str_in_output,
            ignore_eos=ignore_eos,
            max_tokens=max_tokens,
            min_tokens=min_tokens,
            logprobs=logprobs,
            prompt_logprobs=prompt_logprobs,
            detokenize=detokenize,
            skip_special_tokens=skip_special_tokens,
            spaces_between_special_tokens=spaces_between_special_tokens,
            logits_processors=logits_processors,
            truncate_prompt_tokens=truncate_prompt_tokens,
290
            output_kind=output_kind,
291
292
293
            guided_decoding=guided_decoding,
            logit_bias=logit_bias,
            allowed_token_ids=allowed_token_ids,
294
            extra_args=extra_args,
295
296
        )

297
    def __post_init__(self) -> None:
298
299
300
301
302
303
304
305
306
307
308
309
310
311
        # how we deal with `best_of``:
        # if `best_of`` is not set, we default to `n`;
        # if `best_of`` is set, we set `n`` to `best_of`,
        # and set `_real_n`` to the original `n`.
        # when we return the result, we will check
        # if we need to return `n` or `_real_n` results
        if self.best_of:
            if self.best_of < self.n:
                raise ValueError(
                    f"best_of must be greater than or equal to n, "
                    f"got n={self.n} and best_of={self.best_of}.")
            if not self._real_n:
                self._real_n = self.n
                self.n = self.best_of
312

313
        if 0 < self.temperature < _MAX_TEMP:
314
315
316
            logger.warning(
                "temperature %s is less than %s, which may cause numerical "
                "errors nan or inf in tensors. We have maxed it out to %s.",
317
318
                self.temperature, _MAX_TEMP, _MAX_TEMP)
            self.temperature = max(self.temperature, _MAX_TEMP)
319

320
        if self.seed == -1:
321
            self.seed = None
322

323
        if self.stop is None:
324
            self.stop = []
325
326
        elif isinstance(self.stop, str):
            self.stop = [self.stop]
327

328
        if self.stop_token_ids is None:
329
            self.stop_token_ids = []
330
331
332
333

        if self.bad_words is None:
            self.bad_words = []

334
335
336
337
338
        if self.logprobs is True:
            self.logprobs = 1

        if self.prompt_logprobs is True:
            self.prompt_logprobs = 1
339

340
341
        # Number of characters to hold back for stop string evaluation
        # until sequence is finished.
342
        if self.stop and not self.include_stop_str_in_output:
343
344
            self.output_text_buffer_length = max(len(s) for s in self.stop) - 1

345
        self._verify_args()
346
347
348
349

        if self.temperature < _SAMPLING_EPS:
            # Zero temperature means greedy sampling.
            self.top_p = 1.0
350
            self.top_k = 0
351
352
            self.min_p = 0.0
            self._verify_greedy_sampling()
353

354
        # eos_token_id is added to this by the engine
355
        self._all_stop_token_ids.update(self.stop_token_ids)
356
357

    def _verify_args(self) -> None:
358
359
360
        if not isinstance(self.n, int):
            raise ValueError(f"n must be an int, but is of "
                             f"type {type(self.n)}")
361
362
        if self.n < 1:
            raise ValueError(f"n must be at least 1, got {self.n}.")
363
364
365
366
367
368
369
370
371
372
373
        if self.best_of is not None:
            if not isinstance(self.best_of, int):
                raise ValueError(
                    f"best_of must be an integer, got {type(self.best_of)}")
            if self.best_of < 1:
                raise ValueError(
                    f"best_of must be at least 1, got {self.best_of}")
            if self.best_of < self.n:
                raise ValueError(
                    f"best_of must be greater than or equal to n, "
                    f"got n={self.n} and best_of={self.best_of}.")
374
375
376
377
378
379
        if not -2.0 <= self.presence_penalty <= 2.0:
            raise ValueError("presence_penalty must be in [-2, 2], got "
                             f"{self.presence_penalty}.")
        if not -2.0 <= self.frequency_penalty <= 2.0:
            raise ValueError("frequency_penalty must be in [-2, 2], got "
                             f"{self.frequency_penalty}.")
380
381
382
383
        if self.repetition_penalty <= 0.0:
            raise ValueError(
                "repetition_penalty must be greater than zero, got "
                f"{self.repetition_penalty}.")
384
385
386
387
388
        if self.temperature < 0.0:
            raise ValueError(
                f"temperature must be non-negative, got {self.temperature}.")
        if not 0.0 < self.top_p <= 1.0:
            raise ValueError(f"top_p must be in (0, 1], got {self.top_p}.")
389
390
391
        # quietly accept -1 as disabled, but prefer 0
        if self.top_k < -1:
            raise ValueError(f"top_k must be 0 (disable), or at least 1, "
392
                             f"got {self.top_k}.")
393
394
395
        if not isinstance(self.top_k, int):
            raise TypeError(
                f"top_k must be an integer, got {type(self.top_k).__name__}")
Roy's avatar
Roy committed
396
397
398
        if not 0.0 <= self.min_p <= 1.0:
            raise ValueError("min_p must be in [0, 1], got "
                             f"{self.min_p}.")
399
        if self.max_tokens is not None and self.max_tokens < 1:
400
401
            raise ValueError(
                f"max_tokens must be at least 1, got {self.max_tokens}.")
402
403
404
405
406
407
408
        if self.min_tokens < 0:
            raise ValueError(f"min_tokens must be greater than or equal to 0, "
                             f"got {self.min_tokens}.")
        if self.max_tokens is not None and self.min_tokens > self.max_tokens:
            raise ValueError(
                f"min_tokens must be less than or equal to "
                f"max_tokens={self.max_tokens}, got {self.min_tokens}.")
409
410
        if (self.logprobs is not None and self.logprobs != -1
                and self.logprobs < 0):
411
            raise ValueError(
412
                f"logprobs must be non-negative or -1, got {self.logprobs}.")
413
414
415
416
417
        if (self.prompt_logprobs is not None and self.prompt_logprobs != -1
                and self.prompt_logprobs < 0):
            raise ValueError(
                f"prompt_logprobs must be non-negative or -1, got "
                f"{self.prompt_logprobs}.")
418
        if (self.truncate_prompt_tokens is not None
419
420
421
422
423
                and (self.truncate_prompt_tokens == 0
                     or self.truncate_prompt_tokens < -1)):
            raise ValueError(
                f"truncate_prompt_tokens must be an integer >= 1 or -1, "
                f"got {self.truncate_prompt_tokens}")
424
425
426
427
        assert isinstance(self.stop_token_ids, list)
        if not all(isinstance(st_id, int) for st_id in self.stop_token_ids):
            raise ValueError(f"stop_token_ids must contain only integers, "
                             f"got {self.stop_token_ids}.")
428
        assert isinstance(self.stop, list)
429
430
        if any(not stop_str for stop_str in self.stop):
            raise ValueError("stop cannot contain an empty string.")
431
432
433
434
        if self.stop and not self.detokenize:
            raise ValueError(
                "stop strings are only supported when detokenize is True. "
                "Set detokenize=True to use stop.")
435
436
437
        if self.best_of != self._real_n and self.output_kind == (
                RequestOutputKind.DELTA):
            raise ValueError("best_of must equal n to use output_kind=DELTA")
438
439

    def _verify_greedy_sampling(self) -> None:
440
441
442
        if self.n > 1:
            raise ValueError("n must be 1 when using greedy sampling, "
                             f"got {self.n}.")
443

444
    def update_from_generation_config(
445
            self,
446
            generation_config: dict[str, Any],
447
            model_eos_token_id: Optional[int] = None) -> None:
448
        """Update if there are non-default values from generation_config"""
449
450
451
452

        if model_eos_token_id is not None:
            # Add the eos token id into the sampling_params to support
            # min_tokens processing.
453
            self._all_stop_token_ids.add(model_eos_token_id)
454

455
        # Update eos_token_id for generation
456
        if (eos_ids := generation_config.get("eos_token_id")) is not None:
457
            # it can be either int or list of int
458
459
460
461
462
463
464
            eos_ids = {eos_ids} if isinstance(eos_ids, int) else set(eos_ids)
            if model_eos_token_id is not None:
                # We don't need to include the primary eos_token_id in
                # stop_token_ids since it's handled separately for stopping
                # purposes.
                eos_ids.discard(model_eos_token_id)
            if eos_ids:
465
                self._all_stop_token_ids.update(eos_ids)
466
467
468
                if not self.ignore_eos:
                    eos_ids.update(self.stop_token_ids)
                    self.stop_token_ids = list(eos_ids)
469

470
    def update_from_tokenizer(self, tokenizer: AnyTokenizer) -> None:
471
        if not self.bad_words:
472
            return
473
        self._bad_words_token_ids = []
474
475
476
477
478
479
480
        for bad_word in self.bad_words:
            # To prohibit words both at the beginning
            # and in the middle of text
            # (related to add_prefix_space tokenizer parameter)
            for add_prefix_space in [False, True]:
                prefix = " " if add_prefix_space else ""
                prompt = prefix + bad_word.lstrip()
481
482
                prompt_token_ids = tokenizer.encode(text=prompt,
                                                    add_special_tokens=False)
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505

                # If no space at the beginning
                # or if prefix space produces a new word token
                if (not add_prefix_space) or (
                        add_prefix_space and prompt_token_ids[0]
                        != self._bad_words_token_ids[-1][0]
                        and len(prompt_token_ids) == len(
                            self._bad_words_token_ids[-1])):
                    self._bad_words_token_ids.append(prompt_token_ids)

        invalid_token_ids = [
            token_id for bad_words_token_ids in self._bad_words_token_ids
            for token_id in bad_words_token_ids
            if token_id < 0 or token_id > tokenizer.max_token_id
        ]
        if len(invalid_token_ids) > 0:
            raise ValueError(
                f"The model vocabulary size is {tokenizer.max_token_id+1},"
                f" but the following tokens"
                f" were specified as bad: {invalid_token_ids}."
                f" All token id values should be integers satisfying:"
                f" 0 <= token_id <= {tokenizer.max_token_id}.")

506
507
508
509
    @cached_property
    def sampling_type(self) -> SamplingType:
        if self.temperature < _SAMPLING_EPS:
            return SamplingType.GREEDY
Nick Hill's avatar
Nick Hill committed
510
511
        if self.seed is not None:
            return SamplingType.RANDOM_SEED
512
513
        return SamplingType.RANDOM

514
    @property
515
    def all_stop_token_ids(self) -> set[int]:
516
517
        return self._all_stop_token_ids

518
    @property
519
    def bad_words_token_ids(self) -> Optional[list[list[int]]]:
520
521
522
        # For internal use only. Backward compatibility not guaranteed
        return self._bad_words_token_ids

523
    def clone(self) -> "SamplingParams":
524
        """Deep copy, but maybe not the LogitsProcessor objects.
525

526
527
528
        LogitsProcessor objects may contain an arbitrary, nontrivial amount of
        data that is expensive to copy. However, if not copied, the processor
        needs to support parallel decoding for multiple sequences
529
530
531
532
        See https://github.com/vllm-project/vllm/issues/3087
        """

        logit_processor_refs = None if self.logits_processors is None else {
533
            id(lp): lp.clone() if hasattr(lp, 'clone') else lp
534
535
536
537
            for lp in self.logits_processors
        }
        return copy.deepcopy(self, memo=logit_processor_refs)

538
    def __repr__(self) -> str:
539
540
541
542
543
544
545
546
547
        return (
            f"SamplingParams(n={self.n}, "
            f"presence_penalty={self.presence_penalty}, "
            f"frequency_penalty={self.frequency_penalty}, "
            f"repetition_penalty={self.repetition_penalty}, "
            f"temperature={self.temperature}, "
            f"top_p={self.top_p}, "
            f"top_k={self.top_k}, "
            f"min_p={self.min_p}, "
Nick Hill's avatar
Nick Hill committed
548
            f"seed={self.seed}, "
549
550
            f"stop={self.stop}, "
            f"stop_token_ids={self.stop_token_ids}, "
551
            f"bad_words={self.bad_words}, "
552
553
554
            f"include_stop_str_in_output={self.include_stop_str_in_output}, "
            f"ignore_eos={self.ignore_eos}, "
            f"max_tokens={self.max_tokens}, "
555
            f"min_tokens={self.min_tokens}, "
556
557
558
559
            f"logprobs={self.logprobs}, "
            f"prompt_logprobs={self.prompt_logprobs}, "
            f"skip_special_tokens={self.skip_special_tokens}, "
            "spaces_between_special_tokens="
560
            f"{self.spaces_between_special_tokens}, "
561
            f"truncate_prompt_tokens={self.truncate_prompt_tokens}, "
562
563
            f"guided_decoding={self.guided_decoding}, "
            f"extra_args={self.extra_args})")
564
565
566
567
568
569
570
571
572
573
574
575


class BeamSearchParams(
        msgspec.Struct,
        omit_defaults=True,  # type: ignore[call-arg]
        # required for @cached_property.
        dict=True):  # type: ignore[call-arg]
    """Beam search parameters for text generation."""
    beam_width: int
    max_tokens: int
    ignore_eos: bool = False
    temperature: float = 0.0
576
    length_penalty: float = 1.0
577
    include_stop_str_in_output: bool = False