chat_completions.rs 12.8 KB
Newer Older
1
2
3
// SPDX-FileCopyrightText: Copyright (c) 2024-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0

4
5
6
7
use dynamo_runtime::protocols::annotated::AnnotationsProvider;
use serde::{Deserialize, Serialize};
use validator::Validate;

8
9
10
use crate::engines::ValidateRequest;

use super::{
11
    OpenAIOutputOptionsProvider, OpenAISamplingOptionsProvider, OpenAIStopConditionsProvider,
12
13
14
    common_ext::{
        CommonExt, CommonExtProvider, choose_with_deprecation, emit_nvext_deprecation_warning,
    },
15
16
    nvext::NvExt,
    nvext::NvExtProvider,
17
    validate,
18
};
19

20
pub mod aggregator;
21
mod delta;
Ryan Olson's avatar
Ryan Olson committed
22
pub mod jail;
23

Paul Hendricks's avatar
Paul Hendricks committed
24
pub use aggregator::DeltaAggregator;
25
26
pub use delta::DeltaGenerator;

27
/// A request structure for creating a chat completion, extending OpenAI's
28
/// `CreateChatCompletionRequest` with [`NvExt`] extensions and common fields.
29
30
31
///
/// # Fields
/// - `inner`: The base OpenAI chat completion request, embedded using `serde(flatten)`.
32
33
34
/// - `common`: Common extension fields (ignore_eos, min_tokens) at root level, embedded using `serde(flatten)`.
/// - `nvext`: The optional NVIDIA extension field. See [`NvExt`] for more details.
///   Note: If ignore_eos is specified in both common and nvext, the common (root-level) value takes precedence.
Paul Hendricks's avatar
Paul Hendricks committed
35
#[derive(Serialize, Deserialize, Validate, Debug, Clone)]
36
pub struct NvCreateChatCompletionRequest {
Paul Hendricks's avatar
Paul Hendricks committed
37
    #[serde(flatten)]
38
    pub inner: dynamo_async_openai::types::CreateChatCompletionRequest,
39

40
41
42
    #[serde(flatten, default)]
    pub common: CommonExt,

43
    #[serde(skip_serializing_if = "Option::is_none")]
44
    pub nvext: Option<NvExt>,
45
46
47
48

    /// Extra args to pass to the chat template rendering context
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub chat_template_args: Option<std::collections::HashMap<String, serde_json::Value>>,
49
50
}

51
52
53
54
55
56
/// A response structure for unary chat completion responses, embedding OpenAI's
/// `CreateChatCompletionResponse`.
///
/// # Fields
/// - `inner`: The base OpenAI unary chat completion response, embedded
///   using `serde(flatten)`.
57
pub type NvCreateChatCompletionResponse = dynamo_async_openai::types::CreateChatCompletionResponse;
58

59
60
61
62
63
64
/// A response structure for streamed chat completions, embedding OpenAI's
/// `CreateChatCompletionStreamResponse`.
///
/// # Fields
/// - `inner`: The base OpenAI streaming chat completion response, embedded
///   using `serde(flatten)`.
65
66
pub type NvCreateChatCompletionStreamResponse =
    dynamo_async_openai::types::CreateChatCompletionStreamResponse;
67

68
69
/// Implements `NvExtProvider` for `NvCreateChatCompletionRequest`,
/// providing access to NVIDIA-specific extensions.
70
impl NvExtProvider for NvCreateChatCompletionRequest {
71
    /// Returns a reference to the optional `NvExt` extension, if available.
72
73
74
75
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }

76
    /// Returns `None`, as raw prompt extraction is not implemented.
77
78
79
80
81
    fn raw_prompt(&self) -> Option<String> {
        None
    }
}

82
83
/// Implements `AnnotationsProvider` for `NvCreateChatCompletionRequest`,
/// enabling retrieval and management of request annotations.
84
impl AnnotationsProvider for NvCreateChatCompletionRequest {
85
    /// Retrieves the list of annotations from `NvExt`, if present.
Biswa Panda's avatar
Biswa Panda committed
86
87
88
89
90
91
    fn annotations(&self) -> Option<Vec<String>> {
        self.nvext
            .as_ref()
            .and_then(|nvext| nvext.annotations.clone())
    }

92
93
94
95
96
97
98
    /// Checks whether a specific annotation exists in the request.
    ///
    /// # Arguments
    /// * `annotation` - A string slice representing the annotation to check.
    ///
    /// # Returns
    /// `true` if the annotation exists, `false` otherwise.
Biswa Panda's avatar
Biswa Panda committed
99
100
101
102
103
104
105
106
    fn has_annotation(&self, annotation: &str) -> bool {
        self.nvext
            .as_ref()
            .and_then(|nvext| nvext.annotations.as_ref())
            .map(|annotations| annotations.contains(&annotation.to_string()))
            .unwrap_or(false)
    }
}
107

108
109
/// Implements `OpenAISamplingOptionsProvider` for `NvCreateChatCompletionRequest`,
/// exposing OpenAI's sampling parameters for chat completion.
110
impl OpenAISamplingOptionsProvider for NvCreateChatCompletionRequest {
111
    /// Retrieves the temperature parameter for sampling, if set.
112
    fn get_temperature(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
113
        self.inner.temperature
114
115
    }

116
    /// Retrieves the top-p (nucleus sampling) parameter, if set.
117
    fn get_top_p(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
118
        self.inner.top_p
119
120
    }

121
    /// Retrieves the frequency penalty parameter, if set.
122
    fn get_frequency_penalty(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
123
        self.inner.frequency_penalty
124
125
    }

126
    /// Retrieves the presence penalty parameter, if set.
127
    fn get_presence_penalty(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
128
        self.inner.presence_penalty
129
130
    }

131
    /// Returns a reference to the optional `NvExt` extension, if available.
132
133
134
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }
135
136
137
138
139
140
141
142
143
144
145
146
147
148
    /// Retrieves the seed value for random number generation, if set.
    fn get_seed(&self) -> Option<i64> {
        self.inner.seed
    }

    /// Retrieves the number of completions to generate for each prompt, if set.
    fn get_n(&self) -> Option<u8> {
        self.inner.n
    }

    /// Retrieves the best_of parameter, if set.
    fn get_best_of(&self) -> Option<u8> {
        None // Not supported in chat completions
    }
149
150
}

151
152
153
154
155
156
157
158
159
160
/// Implements `CommonExtProvider` for `NvCreateChatCompletionRequest`,
/// providing access to common extension fields.
impl CommonExtProvider for NvCreateChatCompletionRequest {
    /// Returns a reference to the CommonExt struct.
    fn common_ext(&self) -> Option<&CommonExt> {
        Some(&self.common)
    }

    /// Guided Decoding Options
    fn get_guided_json(&self) -> Option<&serde_json::Value> {
161
162
163
164
165
166
        // Note: This one needs special handling since it returns a reference
        if let Some(nvext) = &self.nvext
            && nvext.guided_json.is_some()
        {
            emit_nvext_deprecation_warning("guided_json", true, self.common.guided_json.is_some());
        }
167
168
169
170
171
172
173
        self.common
            .guided_json
            .as_ref()
            .or_else(|| self.nvext.as_ref().and_then(|nv| nv.guided_json.as_ref()))
    }

    fn get_guided_regex(&self) -> Option<String> {
174
175
176
177
178
        choose_with_deprecation(
            "guided_regex",
            self.common.guided_regex.as_ref(),
            self.nvext.as_ref().and_then(|nv| nv.guided_regex.as_ref()),
        )
179
180
181
    }

    fn get_guided_grammar(&self) -> Option<String> {
182
183
184
185
186
187
188
        choose_with_deprecation(
            "guided_grammar",
            self.common.guided_grammar.as_ref(),
            self.nvext
                .as_ref()
                .and_then(|nv| nv.guided_grammar.as_ref()),
        )
189
190
191
    }

    fn get_guided_choice(&self) -> Option<Vec<String>> {
192
193
194
195
196
        choose_with_deprecation(
            "guided_choice",
            self.common.guided_choice.as_ref(),
            self.nvext.as_ref().and_then(|nv| nv.guided_choice.as_ref()),
        )
197
198
199
    }

    fn get_guided_decoding_backend(&self) -> Option<String> {
200
201
202
        choose_with_deprecation(
            "guided_decoding_backend",
            self.common.guided_decoding_backend.as_ref(),
203
204
            self.nvext
                .as_ref()
205
206
                .and_then(|nv| nv.guided_decoding_backend.as_ref()),
        )
207
    }
208

209
210
211
212
213
214
215
216
217
218
    fn get_guided_whitespace_pattern(&self) -> Option<String> {
        choose_with_deprecation(
            "guided_whitespace_pattern",
            self.common.guided_whitespace_pattern.as_ref(),
            self.nvext
                .as_ref()
                .and_then(|nv| nv.guided_whitespace_pattern.as_ref()),
        )
    }

219
220
221
222
223
224
225
226
    fn get_top_k(&self) -> Option<i32> {
        choose_with_deprecation(
            "top_k",
            self.common.top_k.as_ref(),
            self.nvext.as_ref().and_then(|nv| nv.top_k.as_ref()),
        )
    }

227
228
229
230
231
232
233
234
    fn get_min_p(&self) -> Option<f32> {
        choose_with_deprecation(
            "min_p",
            self.common.min_p.as_ref(),
            self.nvext.as_ref().and_then(|nv| nv.min_p.as_ref()),
        )
    }

235
236
237
238
239
240
241
242
243
    fn get_repetition_penalty(&self) -> Option<f32> {
        choose_with_deprecation(
            "repetition_penalty",
            self.common.repetition_penalty.as_ref(),
            self.nvext
                .as_ref()
                .and_then(|nv| nv.repetition_penalty.as_ref()),
        )
    }
244
245
246
247

    fn get_include_stop_str_in_output(&self) -> Option<bool> {
        self.common.include_stop_str_in_output
    }
248
249
}

250
251
/// Implements `OpenAIStopConditionsProvider` for `NvCreateChatCompletionRequest`,
/// providing access to stop conditions that control chat completion behavior.
252
impl OpenAIStopConditionsProvider for NvCreateChatCompletionRequest {
253
    /// Retrieves the maximum number of tokens allowed in the response.
254
    #[allow(deprecated)]
Paul Hendricks's avatar
Paul Hendricks committed
255
    fn get_max_tokens(&self) -> Option<u32> {
256
        self.inner.max_completion_tokens.or(self.inner.max_tokens)
257
258
    }

259
    /// Retrieves the minimum number of tokens required in the response.
260
261
    /// Returns `min_tokens` Value
    /// `min_tokens` is not an OpenAI-supported parameter.
Paul Hendricks's avatar
Paul Hendricks committed
262
    fn get_min_tokens(&self) -> Option<u32> {
263
        self.common.min_tokens
264
265
    }

266
267
268
269
270
271
272
    /// Retrieves the stop conditions that terminate the chat completion response.
    ///
    /// Converts OpenAI's `Stop` enum to a `Vec<String>`, normalizing the representation.
    ///
    /// # Returns
    /// * `Some(Vec<String>)` if stop conditions are set.
    /// * `None` if no stop conditions are defined.
273
    fn get_stop(&self) -> Option<Vec<String>> {
274
        self.inner.stop.as_ref().map(|stop| match stop {
275
276
            dynamo_async_openai::types::Stop::String(s) => vec![s.clone()],
            dynamo_async_openai::types::Stop::StringArray(arr) => arr.clone(),
277
        })
278
279
    }

280
    /// Returns a reference to the optional `NvExt` extension, if available.
281
282
283
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }
284
285
286
287
288

    /// Get ignore_eos from CommonExt.
    fn get_common_ignore_eos(&self) -> Option<bool> {
        self.common.ignore_eos
    }
289
290
291
292
293
294
295
296
297
298

    /// Get the effective ignore_eos value, considering both CommonExt and NvExt.
    /// CommonExt (root-level) takes precedence over NvExt.
    fn get_ignore_eos(&self) -> Option<bool> {
        choose_with_deprecation(
            "ignore_eos",
            self.get_common_ignore_eos().as_ref(),
            NvExtProvider::nvext(self).and_then(|nv| nv.ignore_eos.as_ref()),
        )
    }
299
}
300

Greg Clark's avatar
Greg Clark committed
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
impl OpenAIOutputOptionsProvider for NvCreateChatCompletionRequest {
    fn get_logprobs(&self) -> Option<u32> {
        match self.inner.logprobs {
            Some(true) => match self.inner.top_logprobs {
                Some(top_logprobs) => Some(top_logprobs as u32),
                None => Some(1_u32),
            },
            Some(false) => None,
            None => None,
        }
    }

    fn get_prompt_logprobs(&self) -> Option<u32> {
        None
    }

    fn get_skip_special_tokens(&self) -> Option<bool> {
        None
    }

    fn get_formatted_prompt(&self) -> Option<bool> {
        None
    }
}

326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
/// Implements `ValidateRequest` for `NvCreateChatCompletionRequest`,
/// allowing us to validate the data.
impl ValidateRequest for NvCreateChatCompletionRequest {
    fn validate(&self) -> Result<(), anyhow::Error> {
        validate::validate_messages(&self.inner.messages)?;
        validate::validate_model(&self.inner.model)?;
        // none for store
        validate::validate_reasoning_effort(&self.inner.reasoning_effort)?;
        validate::validate_metadata(&self.inner.metadata)?;
        validate::validate_frequency_penalty(self.inner.frequency_penalty)?;
        validate::validate_logit_bias(&self.inner.logit_bias)?;
        // none for logprobs
        validate::validate_top_logprobs(self.inner.top_logprobs)?;
        // validate::validate_max_tokens(self.inner.max_tokens)?; // warning depricated field
        validate::validate_max_completion_tokens(self.inner.max_completion_tokens)?;
        validate::validate_n(self.inner.n)?;
        // none for modalities
        // none for prediction
        // none for audio
        validate::validate_presence_penalty(self.inner.presence_penalty)?;
        // none for response_format
        // none for seed
        validate::validate_service_tier(&self.inner.service_tier)?;
        validate::validate_stop(&self.inner.stop)?;
        // none for stream
        // none for stream_options
        validate::validate_temperature(self.inner.temperature)?;
        validate::validate_top_p(self.inner.top_p)?;
        validate::validate_tools(&self.inner.tools.as_deref())?;
        // none for tool_choice
        // none for parallel_tool_calls
        validate::validate_user(self.inner.user.as_deref())?;
        // none for function call
        // none for functions
360
361
        // Common Ext
        validate::validate_repetition_penalty(self.get_repetition_penalty())?;
362
363
        validate::validate_min_p(self.get_min_p())?;
        validate::validate_top_k(self.get_top_k())?;
364
365
366
367

        Ok(())
    }
}