chat_completions.rs 13.7 KB
Newer Older
1
// SPDX-FileCopyrightText: Copyright (c) 2024-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
2
3
// SPDX-License-Identifier: Apache-2.0

4
5
use dynamo_runtime::protocols::annotated::AnnotationsProvider;
use serde::{Deserialize, Serialize};
6
use utoipa::ToSchema;
7
8
use validator::Validate;

9
use crate::engines::ValidateRequest;
10
use crate::preprocessor::media::MediaDecoder;
11
12

use super::{
13
    OpenAIOutputOptionsProvider, OpenAISamplingOptionsProvider, OpenAIStopConditionsProvider,
14
    common_ext::{CommonExt, CommonExtProvider},
15
16
    nvext::NvExt,
    nvext::NvExtProvider,
17
    tools, validate,
18
};
19

20
pub mod aggregator;
21
mod delta;
Ryan Olson's avatar
Ryan Olson committed
22
pub mod jail;
23

Paul Hendricks's avatar
Paul Hendricks committed
24
pub use aggregator::DeltaAggregator;
25
26
pub use delta::DeltaGenerator;

27
/// A request structure for creating a chat completion, extending OpenAI's
28
/// `CreateChatCompletionRequest` with [`NvExt`] extensions and common fields.
29
30
31
///
/// # Fields
/// - `inner`: The base OpenAI chat completion request, embedded using `serde(flatten)`.
32
33
34
/// - `common`: Common extension fields (ignore_eos, min_tokens) at root level, embedded using `serde(flatten)`.
/// - `nvext`: The optional NVIDIA extension field. See [`NvExt`] for more details.
///   Note: If ignore_eos is specified in both common and nvext, the common (root-level) value takes precedence.
35
#[derive(ToSchema, Serialize, Deserialize, Validate, Debug, Clone)]
36
pub struct NvCreateChatCompletionRequest {
Paul Hendricks's avatar
Paul Hendricks committed
37
    #[serde(flatten)]
38
    pub inner: dynamo_async_openai::types::CreateChatCompletionRequest,
39

40
41
42
    #[serde(flatten, default)]
    pub common: CommonExt,

43
    #[serde(skip_serializing_if = "Option::is_none")]
44
    pub nvext: Option<NvExt>,
45
46

    /// Extra args to pass to the chat template rendering context
47
48
49
50
51
52
    /// Also accepts "chat_template_kwargs" as an alias for compatibility
    #[serde(
        default,
        skip_serializing_if = "Option::is_none",
        alias = "chat_template_kwargs"
    )]
53
    pub chat_template_args: Option<std::collections::HashMap<String, serde_json::Value>>,
54

55
56
57
58
59
60
    /// Runtime media decoding parameters.
    /// When provided, these override the MDC defaults
    /// Example: `{"video": {"num_frames": 16}}`
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub media_io_kwargs: Option<MediaDecoder>,

61
62
63
    /// Catch-all for unsupported fields - checked during validation
    #[serde(flatten, default, skip_serializing)]
    pub unsupported_fields: std::collections::HashMap<String, serde_json::Value>,
64
65
}

66
67
68
69
70
71
/// A response structure for unary chat completion responses, embedding OpenAI's
/// `CreateChatCompletionResponse`.
///
/// # Fields
/// - `inner`: The base OpenAI unary chat completion response, embedded
///   using `serde(flatten)`.
72
pub type NvCreateChatCompletionResponse = dynamo_async_openai::types::CreateChatCompletionResponse;
73

74
75
76
77
78
79
/// A response structure for streamed chat completions, embedding OpenAI's
/// `CreateChatCompletionStreamResponse`.
///
/// # Fields
/// - `inner`: The base OpenAI streaming chat completion response, embedded
///   using `serde(flatten)`.
80
81
pub type NvCreateChatCompletionStreamResponse =
    dynamo_async_openai::types::CreateChatCompletionStreamResponse;
82

83
84
/// Implements `NvExtProvider` for `NvCreateChatCompletionRequest`,
/// providing access to NVIDIA-specific extensions.
85
impl NvExtProvider for NvCreateChatCompletionRequest {
86
    /// Returns a reference to the optional `NvExt` extension, if available.
87
88
89
90
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }

91
    /// Returns `None`, as raw prompt extraction is not implemented.
92
93
94
95
96
    fn raw_prompt(&self) -> Option<String> {
        None
    }
}

97
98
/// Implements `AnnotationsProvider` for `NvCreateChatCompletionRequest`,
/// enabling retrieval and management of request annotations.
99
impl AnnotationsProvider for NvCreateChatCompletionRequest {
100
    /// Retrieves the list of annotations from `NvExt`, if present.
Biswa Panda's avatar
Biswa Panda committed
101
102
103
104
105
106
    fn annotations(&self) -> Option<Vec<String>> {
        self.nvext
            .as_ref()
            .and_then(|nvext| nvext.annotations.clone())
    }

107
108
109
110
111
112
113
    /// Checks whether a specific annotation exists in the request.
    ///
    /// # Arguments
    /// * `annotation` - A string slice representing the annotation to check.
    ///
    /// # Returns
    /// `true` if the annotation exists, `false` otherwise.
Biswa Panda's avatar
Biswa Panda committed
114
115
116
117
118
119
120
121
    fn has_annotation(&self, annotation: &str) -> bool {
        self.nvext
            .as_ref()
            .and_then(|nvext| nvext.annotations.as_ref())
            .map(|annotations| annotations.contains(&annotation.to_string()))
            .unwrap_or(false)
    }
}
122

123
124
/// Implements `OpenAISamplingOptionsProvider` for `NvCreateChatCompletionRequest`,
/// exposing OpenAI's sampling parameters for chat completion.
125
impl OpenAISamplingOptionsProvider for NvCreateChatCompletionRequest {
126
    /// Retrieves the temperature parameter for sampling, if set.
127
    fn get_temperature(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
128
        self.inner.temperature
129
130
    }

131
    /// Retrieves the top-p (nucleus sampling) parameter, if set.
132
    fn get_top_p(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
133
        self.inner.top_p
134
135
    }

136
    /// Retrieves the frequency penalty parameter, if set.
137
    fn get_frequency_penalty(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
138
        self.inner.frequency_penalty
139
140
    }

141
    /// Retrieves the presence penalty parameter, if set.
142
    fn get_presence_penalty(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
143
        self.inner.presence_penalty
144
145
    }

146
    /// Returns a reference to the optional `NvExt` extension, if available.
147
148
149
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }
150
151
152
153
154
155
156
157
158
159
160
161
162
163
    /// Retrieves the seed value for random number generation, if set.
    fn get_seed(&self) -> Option<i64> {
        self.inner.seed
    }

    /// Retrieves the number of completions to generate for each prompt, if set.
    fn get_n(&self) -> Option<u8> {
        self.inner.n
    }

    /// Retrieves the best_of parameter, if set.
    fn get_best_of(&self) -> Option<u8> {
        None // Not supported in chat completions
    }
164
165
}

166
167
168
169
170
171
172
173
174
/// Implements `CommonExtProvider` for `NvCreateChatCompletionRequest`,
/// providing access to common extension fields.
impl CommonExtProvider for NvCreateChatCompletionRequest {
    /// Returns a reference to the CommonExt struct.
    fn common_ext(&self) -> Option<&CommonExt> {
        Some(&self.common)
    }

    /// Guided Decoding Options
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
    fn get_guided_json(&self) -> Option<serde_json::Value> {
        if let Some(value) = self.common.guided_json.clone() {
            return Some(value);
        }

        let tool_choice = self.inner.tool_choice.as_ref()?;
        let tools = self.inner.tools.as_deref()?;

        match tools::get_json_schema_from_tools(Some(tool_choice), Some(tools)) {
            Ok(schema) => schema,
            Err(err) => {
                tracing::warn!(
                    error = %err,
                    "failed to derive guided_json from tool_choice"
                );
                None
            }
        }
193
194
195
    }

    fn get_guided_regex(&self) -> Option<String> {
196
        self.common.guided_regex.clone()
197
198
199
    }

    fn get_guided_grammar(&self) -> Option<String> {
200
        self.common.guided_grammar.clone()
201
202
203
    }

    fn get_guided_choice(&self) -> Option<Vec<String>> {
204
        self.common.guided_choice.clone()
205
206
207
    }

    fn get_guided_decoding_backend(&self) -> Option<String> {
208
        self.common.guided_decoding_backend.clone()
209
    }
210

211
    fn get_guided_whitespace_pattern(&self) -> Option<String> {
212
        self.common.guided_whitespace_pattern.clone()
213
214
    }

215
    fn get_top_k(&self) -> Option<i32> {
216
        self.common.top_k
217
218
    }

219
    fn get_min_p(&self) -> Option<f32> {
220
        self.common.min_p
221
222
    }

223
    fn get_repetition_penalty(&self) -> Option<f32> {
224
        self.common.repetition_penalty
225
    }
226
227
228
229

    fn get_include_stop_str_in_output(&self) -> Option<bool> {
        self.common.include_stop_str_in_output
    }
230
231
232
233

    fn get_skip_special_tokens(&self) -> Option<bool> {
        self.common.skip_special_tokens
    }
234
235
}

236
237
/// Implements `OpenAIStopConditionsProvider` for `NvCreateChatCompletionRequest`,
/// providing access to stop conditions that control chat completion behavior.
238
impl OpenAIStopConditionsProvider for NvCreateChatCompletionRequest {
239
    /// Retrieves the maximum number of tokens allowed in the response.
240
    #[allow(deprecated)]
Paul Hendricks's avatar
Paul Hendricks committed
241
    fn get_max_tokens(&self) -> Option<u32> {
242
        self.inner.max_completion_tokens.or(self.inner.max_tokens)
243
244
    }

245
    /// Retrieves the minimum number of tokens required in the response.
246
247
    /// Returns `min_tokens` Value
    /// `min_tokens` is not an OpenAI-supported parameter.
Paul Hendricks's avatar
Paul Hendricks committed
248
    fn get_min_tokens(&self) -> Option<u32> {
249
        self.common.min_tokens
250
251
    }

252
253
254
255
256
257
258
    /// Retrieves the stop conditions that terminate the chat completion response.
    ///
    /// Converts OpenAI's `Stop` enum to a `Vec<String>`, normalizing the representation.
    ///
    /// # Returns
    /// * `Some(Vec<String>)` if stop conditions are set.
    /// * `None` if no stop conditions are defined.
259
    fn get_stop(&self) -> Option<Vec<String>> {
260
        self.inner.stop.as_ref().map(|stop| match stop {
261
262
            dynamo_async_openai::types::Stop::String(s) => vec![s.clone()],
            dynamo_async_openai::types::Stop::StringArray(arr) => arr.clone(),
263
        })
264
265
    }

266
    /// Returns a reference to the optional `NvExt` extension, if available.
267
268
269
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }
270
271
272
273
274

    /// Get ignore_eos from CommonExt.
    fn get_common_ignore_eos(&self) -> Option<bool> {
        self.common.ignore_eos
    }
275

276
    /// Get the effective ignore_eos value from CommonExt.
277
    fn get_ignore_eos(&self) -> Option<bool> {
278
        self.common.ignore_eos
279
    }
280
}
281

Greg Clark's avatar
Greg Clark committed
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
impl OpenAIOutputOptionsProvider for NvCreateChatCompletionRequest {
    fn get_logprobs(&self) -> Option<u32> {
        match self.inner.logprobs {
            Some(true) => match self.inner.top_logprobs {
                Some(top_logprobs) => Some(top_logprobs as u32),
                None => Some(1_u32),
            },
            Some(false) => None,
            None => None,
        }
    }

    fn get_prompt_logprobs(&self) -> Option<u32> {
        None
    }

    fn get_skip_special_tokens(&self) -> Option<bool> {
299
        CommonExtProvider::get_skip_special_tokens(self)
Greg Clark's avatar
Greg Clark committed
300
301
302
303
304
305
306
    }

    fn get_formatted_prompt(&self) -> Option<bool> {
        None
    }
}

307
308
309
310
/// Implements `ValidateRequest` for `NvCreateChatCompletionRequest`,
/// allowing us to validate the data.
impl ValidateRequest for NvCreateChatCompletionRequest {
    fn validate(&self) -> Result<(), anyhow::Error> {
311
        validate::validate_no_unsupported_fields(&self.unsupported_fields)?;
312
313
314
315
        validate::validate_messages(&self.inner.messages)?;
        validate::validate_model(&self.inner.model)?;
        // none for store
        validate::validate_reasoning_effort(&self.inner.reasoning_effort)?;
316
        // none for metadata
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
        validate::validate_frequency_penalty(self.inner.frequency_penalty)?;
        validate::validate_logit_bias(&self.inner.logit_bias)?;
        // none for logprobs
        validate::validate_top_logprobs(self.inner.top_logprobs)?;
        // validate::validate_max_tokens(self.inner.max_tokens)?; // warning depricated field
        validate::validate_max_completion_tokens(self.inner.max_completion_tokens)?;
        validate::validate_n(self.inner.n)?;
        // none for modalities
        // none for prediction
        // none for audio
        validate::validate_presence_penalty(self.inner.presence_penalty)?;
        // none for response_format
        // none for seed
        validate::validate_service_tier(&self.inner.service_tier)?;
        validate::validate_stop(&self.inner.stop)?;
        // none for stream
        // none for stream_options
        validate::validate_temperature(self.inner.temperature)?;
        validate::validate_top_p(self.inner.top_p)?;
        validate::validate_tools(&self.inner.tools.as_deref())?;
        // none for tool_choice
        // none for parallel_tool_calls
        validate::validate_user(self.inner.user.as_deref())?;
        // none for function call
        // none for functions
342
343
        // Common Ext
        validate::validate_repetition_penalty(self.get_repetition_penalty())?;
344
345
        validate::validate_min_p(self.get_min_p())?;
        validate::validate_top_k(self.get_top_k())?;
346
347
        // Cross-field validation
        validate::validate_n_with_temperature(self.inner.n, self.inner.temperature)?;
348
349
350
351

        Ok(())
    }
}
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401

#[cfg(test)]
mod tests {
    use super::*;
    use crate::protocols::common::OutputOptionsProvider;
    use serde_json::json;

    #[test]
    fn test_skip_special_tokens_none() {
        let json_str = json!({
            "model": "test-model",
            "messages": [
                {"role": "user", "content": "Hello"}
            ]
        });

        let request: NvCreateChatCompletionRequest =
            serde_json::from_value(json_str).expect("Failed to deserialize request");

        assert_eq!(request.common.skip_special_tokens, None);

        let output_options = request
            .extract_output_options()
            .expect("Failed to extract output options");

        assert_eq!(output_options.skip_special_tokens, None);
    }

    #[test]
    fn test_skip_special_tokens_propagates() {
        for skip_value in [true, false] {
            let json_str = json!({
                "model": "test-model",
                "messages": [
                    {"role": "user", "content": "Hello"}
                ],
                "skip_special_tokens": skip_value
            });

            let request: NvCreateChatCompletionRequest =
                serde_json::from_value(json_str).expect("Failed to deserialize request");

            let output_options = request
                .extract_output_options()
                .expect("Failed to extract output options");

            assert_eq!(output_options.skip_special_tokens, Some(skip_value));
        }
    }
}