chat_completions.rs 13.2 KB
Newer Older
1
2
3
// SPDX-FileCopyrightText: Copyright (c) 2024-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0

4
5
6
7
use dynamo_runtime::protocols::annotated::AnnotationsProvider;
use serde::{Deserialize, Serialize};
use validator::Validate;

8
9
10
use crate::engines::ValidateRequest;

use super::{
11
    OpenAIOutputOptionsProvider, OpenAISamplingOptionsProvider, OpenAIStopConditionsProvider,
12
    common_ext::{CommonExt, CommonExtProvider},
13
14
    nvext::NvExt,
    nvext::NvExtProvider,
15
    tools, validate,
16
};
17

18
pub mod aggregator;
19
mod delta;
Ryan Olson's avatar
Ryan Olson committed
20
pub mod jail;
21

Paul Hendricks's avatar
Paul Hendricks committed
22
pub use aggregator::DeltaAggregator;
23
24
pub use delta::DeltaGenerator;

25
/// A request structure for creating a chat completion, extending OpenAI's
26
/// `CreateChatCompletionRequest` with [`NvExt`] extensions and common fields.
27
28
29
///
/// # Fields
/// - `inner`: The base OpenAI chat completion request, embedded using `serde(flatten)`.
30
31
32
/// - `common`: Common extension fields (ignore_eos, min_tokens) at root level, embedded using `serde(flatten)`.
/// - `nvext`: The optional NVIDIA extension field. See [`NvExt`] for more details.
///   Note: If ignore_eos is specified in both common and nvext, the common (root-level) value takes precedence.
Paul Hendricks's avatar
Paul Hendricks committed
33
#[derive(Serialize, Deserialize, Validate, Debug, Clone)]
34
pub struct NvCreateChatCompletionRequest {
Paul Hendricks's avatar
Paul Hendricks committed
35
    #[serde(flatten)]
36
    pub inner: dynamo_async_openai::types::CreateChatCompletionRequest,
37

38
39
40
    #[serde(flatten, default)]
    pub common: CommonExt,

41
    #[serde(skip_serializing_if = "Option::is_none")]
42
    pub nvext: Option<NvExt>,
43
44
45
46

    /// Extra args to pass to the chat template rendering context
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub chat_template_args: Option<std::collections::HashMap<String, serde_json::Value>>,
47
48
49
50

    /// Catch-all for unsupported fields - checked during validation
    #[serde(flatten, default, skip_serializing)]
    pub unsupported_fields: std::collections::HashMap<String, serde_json::Value>,
51
52
}

53
54
55
56
57
58
/// A response structure for unary chat completion responses, embedding OpenAI's
/// `CreateChatCompletionResponse`.
///
/// # Fields
/// - `inner`: The base OpenAI unary chat completion response, embedded
///   using `serde(flatten)`.
59
pub type NvCreateChatCompletionResponse = dynamo_async_openai::types::CreateChatCompletionResponse;
60

61
62
63
64
65
66
/// A response structure for streamed chat completions, embedding OpenAI's
/// `CreateChatCompletionStreamResponse`.
///
/// # Fields
/// - `inner`: The base OpenAI streaming chat completion response, embedded
///   using `serde(flatten)`.
67
68
pub type NvCreateChatCompletionStreamResponse =
    dynamo_async_openai::types::CreateChatCompletionStreamResponse;
69

70
71
/// Implements `NvExtProvider` for `NvCreateChatCompletionRequest`,
/// providing access to NVIDIA-specific extensions.
72
impl NvExtProvider for NvCreateChatCompletionRequest {
73
    /// Returns a reference to the optional `NvExt` extension, if available.
74
75
76
77
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }

78
    /// Returns `None`, as raw prompt extraction is not implemented.
79
80
81
82
83
    fn raw_prompt(&self) -> Option<String> {
        None
    }
}

84
85
/// Implements `AnnotationsProvider` for `NvCreateChatCompletionRequest`,
/// enabling retrieval and management of request annotations.
86
impl AnnotationsProvider for NvCreateChatCompletionRequest {
87
    /// Retrieves the list of annotations from `NvExt`, if present.
Biswa Panda's avatar
Biswa Panda committed
88
89
90
91
92
93
    fn annotations(&self) -> Option<Vec<String>> {
        self.nvext
            .as_ref()
            .and_then(|nvext| nvext.annotations.clone())
    }

94
95
96
97
98
99
100
    /// Checks whether a specific annotation exists in the request.
    ///
    /// # Arguments
    /// * `annotation` - A string slice representing the annotation to check.
    ///
    /// # Returns
    /// `true` if the annotation exists, `false` otherwise.
Biswa Panda's avatar
Biswa Panda committed
101
102
103
104
105
106
107
108
    fn has_annotation(&self, annotation: &str) -> bool {
        self.nvext
            .as_ref()
            .and_then(|nvext| nvext.annotations.as_ref())
            .map(|annotations| annotations.contains(&annotation.to_string()))
            .unwrap_or(false)
    }
}
109

110
111
/// Implements `OpenAISamplingOptionsProvider` for `NvCreateChatCompletionRequest`,
/// exposing OpenAI's sampling parameters for chat completion.
112
impl OpenAISamplingOptionsProvider for NvCreateChatCompletionRequest {
113
    /// Retrieves the temperature parameter for sampling, if set.
114
    fn get_temperature(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
115
        self.inner.temperature
116
117
    }

118
    /// Retrieves the top-p (nucleus sampling) parameter, if set.
119
    fn get_top_p(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
120
        self.inner.top_p
121
122
    }

123
    /// Retrieves the frequency penalty parameter, if set.
124
    fn get_frequency_penalty(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
125
        self.inner.frequency_penalty
126
127
    }

128
    /// Retrieves the presence penalty parameter, if set.
129
    fn get_presence_penalty(&self) -> Option<f32> {
Paul Hendricks's avatar
Paul Hendricks committed
130
        self.inner.presence_penalty
131
132
    }

133
    /// Returns a reference to the optional `NvExt` extension, if available.
134
135
136
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }
137
138
139
140
141
142
143
144
145
146
147
148
149
150
    /// Retrieves the seed value for random number generation, if set.
    fn get_seed(&self) -> Option<i64> {
        self.inner.seed
    }

    /// Retrieves the number of completions to generate for each prompt, if set.
    fn get_n(&self) -> Option<u8> {
        self.inner.n
    }

    /// Retrieves the best_of parameter, if set.
    fn get_best_of(&self) -> Option<u8> {
        None // Not supported in chat completions
    }
151
152
}

153
154
155
156
157
158
159
160
161
/// Implements `CommonExtProvider` for `NvCreateChatCompletionRequest`,
/// providing access to common extension fields.
impl CommonExtProvider for NvCreateChatCompletionRequest {
    /// Returns a reference to the CommonExt struct.
    fn common_ext(&self) -> Option<&CommonExt> {
        Some(&self.common)
    }

    /// Guided Decoding Options
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
    fn get_guided_json(&self) -> Option<serde_json::Value> {
        if let Some(value) = self.common.guided_json.clone() {
            return Some(value);
        }

        let tool_choice = self.inner.tool_choice.as_ref()?;
        let tools = self.inner.tools.as_deref()?;

        match tools::get_json_schema_from_tools(Some(tool_choice), Some(tools)) {
            Ok(schema) => schema,
            Err(err) => {
                tracing::warn!(
                    error = %err,
                    "failed to derive guided_json from tool_choice"
                );
                None
            }
        }
180
181
182
    }

    fn get_guided_regex(&self) -> Option<String> {
183
        self.common.guided_regex.clone()
184
185
186
    }

    fn get_guided_grammar(&self) -> Option<String> {
187
        self.common.guided_grammar.clone()
188
189
190
    }

    fn get_guided_choice(&self) -> Option<Vec<String>> {
191
        self.common.guided_choice.clone()
192
193
194
    }

    fn get_guided_decoding_backend(&self) -> Option<String> {
195
        self.common.guided_decoding_backend.clone()
196
    }
197

198
    fn get_guided_whitespace_pattern(&self) -> Option<String> {
199
        self.common.guided_whitespace_pattern.clone()
200
201
    }

202
    fn get_top_k(&self) -> Option<i32> {
203
        self.common.top_k
204
205
    }

206
    fn get_min_p(&self) -> Option<f32> {
207
        self.common.min_p
208
209
    }

210
    fn get_repetition_penalty(&self) -> Option<f32> {
211
        self.common.repetition_penalty
212
    }
213
214
215
216

    fn get_include_stop_str_in_output(&self) -> Option<bool> {
        self.common.include_stop_str_in_output
    }
217
218
219
220

    fn get_skip_special_tokens(&self) -> Option<bool> {
        self.common.skip_special_tokens
    }
221
222
}

223
224
/// Implements `OpenAIStopConditionsProvider` for `NvCreateChatCompletionRequest`,
/// providing access to stop conditions that control chat completion behavior.
225
impl OpenAIStopConditionsProvider for NvCreateChatCompletionRequest {
226
    /// Retrieves the maximum number of tokens allowed in the response.
227
    #[allow(deprecated)]
Paul Hendricks's avatar
Paul Hendricks committed
228
    fn get_max_tokens(&self) -> Option<u32> {
229
        self.inner.max_completion_tokens.or(self.inner.max_tokens)
230
231
    }

232
    /// Retrieves the minimum number of tokens required in the response.
233
234
    /// Returns `min_tokens` Value
    /// `min_tokens` is not an OpenAI-supported parameter.
Paul Hendricks's avatar
Paul Hendricks committed
235
    fn get_min_tokens(&self) -> Option<u32> {
236
        self.common.min_tokens
237
238
    }

239
240
241
242
243
244
245
    /// Retrieves the stop conditions that terminate the chat completion response.
    ///
    /// Converts OpenAI's `Stop` enum to a `Vec<String>`, normalizing the representation.
    ///
    /// # Returns
    /// * `Some(Vec<String>)` if stop conditions are set.
    /// * `None` if no stop conditions are defined.
246
    fn get_stop(&self) -> Option<Vec<String>> {
247
        self.inner.stop.as_ref().map(|stop| match stop {
248
249
            dynamo_async_openai::types::Stop::String(s) => vec![s.clone()],
            dynamo_async_openai::types::Stop::StringArray(arr) => arr.clone(),
250
        })
251
252
    }

253
    /// Returns a reference to the optional `NvExt` extension, if available.
254
255
256
    fn nvext(&self) -> Option<&NvExt> {
        self.nvext.as_ref()
    }
257
258
259
260
261

    /// Get ignore_eos from CommonExt.
    fn get_common_ignore_eos(&self) -> Option<bool> {
        self.common.ignore_eos
    }
262

263
    /// Get the effective ignore_eos value from CommonExt.
264
    fn get_ignore_eos(&self) -> Option<bool> {
265
        self.common.ignore_eos
266
    }
267
}
268

Greg Clark's avatar
Greg Clark committed
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
impl OpenAIOutputOptionsProvider for NvCreateChatCompletionRequest {
    fn get_logprobs(&self) -> Option<u32> {
        match self.inner.logprobs {
            Some(true) => match self.inner.top_logprobs {
                Some(top_logprobs) => Some(top_logprobs as u32),
                None => Some(1_u32),
            },
            Some(false) => None,
            None => None,
        }
    }

    fn get_prompt_logprobs(&self) -> Option<u32> {
        None
    }

    fn get_skip_special_tokens(&self) -> Option<bool> {
286
        CommonExtProvider::get_skip_special_tokens(self)
Greg Clark's avatar
Greg Clark committed
287
288
289
290
291
292
293
    }

    fn get_formatted_prompt(&self) -> Option<bool> {
        None
    }
}

294
295
296
297
/// Implements `ValidateRequest` for `NvCreateChatCompletionRequest`,
/// allowing us to validate the data.
impl ValidateRequest for NvCreateChatCompletionRequest {
    fn validate(&self) -> Result<(), anyhow::Error> {
298
        validate::validate_no_unsupported_fields(&self.unsupported_fields)?;
299
300
301
302
        validate::validate_messages(&self.inner.messages)?;
        validate::validate_model(&self.inner.model)?;
        // none for store
        validate::validate_reasoning_effort(&self.inner.reasoning_effort)?;
303
        // none for metadata
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
        validate::validate_frequency_penalty(self.inner.frequency_penalty)?;
        validate::validate_logit_bias(&self.inner.logit_bias)?;
        // none for logprobs
        validate::validate_top_logprobs(self.inner.top_logprobs)?;
        // validate::validate_max_tokens(self.inner.max_tokens)?; // warning depricated field
        validate::validate_max_completion_tokens(self.inner.max_completion_tokens)?;
        validate::validate_n(self.inner.n)?;
        // none for modalities
        // none for prediction
        // none for audio
        validate::validate_presence_penalty(self.inner.presence_penalty)?;
        // none for response_format
        // none for seed
        validate::validate_service_tier(&self.inner.service_tier)?;
        validate::validate_stop(&self.inner.stop)?;
        // none for stream
        // none for stream_options
        validate::validate_temperature(self.inner.temperature)?;
        validate::validate_top_p(self.inner.top_p)?;
        validate::validate_tools(&self.inner.tools.as_deref())?;
        // none for tool_choice
        // none for parallel_tool_calls
        validate::validate_user(self.inner.user.as_deref())?;
        // none for function call
        // none for functions
329
330
        // Common Ext
        validate::validate_repetition_penalty(self.get_repetition_penalty())?;
331
332
        validate::validate_min_p(self.get_min_p())?;
        validate::validate_top_k(self.get_top_k())?;
333
334
        // Cross-field validation
        validate::validate_n_with_temperature(self.inner.n, self.inner.temperature)?;
335
336
337
338

        Ok(())
    }
}
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388

#[cfg(test)]
mod tests {
    use super::*;
    use crate::protocols::common::OutputOptionsProvider;
    use serde_json::json;

    #[test]
    fn test_skip_special_tokens_none() {
        let json_str = json!({
            "model": "test-model",
            "messages": [
                {"role": "user", "content": "Hello"}
            ]
        });

        let request: NvCreateChatCompletionRequest =
            serde_json::from_value(json_str).expect("Failed to deserialize request");

        assert_eq!(request.common.skip_special_tokens, None);

        let output_options = request
            .extract_output_options()
            .expect("Failed to extract output options");

        assert_eq!(output_options.skip_special_tokens, None);
    }

    #[test]
    fn test_skip_special_tokens_propagates() {
        for skip_value in [true, false] {
            let json_str = json!({
                "model": "test-model",
                "messages": [
                    {"role": "user", "content": "Hello"}
                ],
                "skip_special_tokens": skip_value
            });

            let request: NvCreateChatCompletionRequest =
                serde_json::from_value(json_str).expect("Failed to deserialize request");

            let output_options = request
                .extract_output_options()
                .expect("Failed to extract output options");

            assert_eq!(output_options.skip_special_tokens, Some(skip_value));
        }
    }
}