Repository navigation
feat(gateway): add MessageRequestBuildingStage for Messages API #744
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -12,6 +12,7 @@ use openai_protocol::{ | |
| chat::ChatCompletionRequest, | ||
| common::{ResponseFormat, StringOrArray, ToolChoice, ToolChoiceValue}, | ||
| generate::GenerateRequest, | ||
| messages::CreateMessageRequest, | ||
| responses::ResponsesRequest, | ||
| sampling_params::SamplingParams as GenerateSamplingParams, | ||
| }; | ||
|
|
@@ -607,6 +608,74 @@ impl SglangSchedulerClient { | |
| } | ||
| } | ||
|
|
||
| /// Build a GenerateRequest from CreateMessageRequest (Anthropic Messages API) | ||
| #[expect( | ||
| clippy::unused_self, | ||
| reason = "method receiver kept for consistent public API" | ||
| )] | ||
| pub fn build_generate_request_from_messages( | ||
| &self, | ||
| request_id: String, | ||
| body: &CreateMessageRequest, | ||
| processed_text: String, | ||
| token_ids: Vec<u32>, | ||
| multimodal_inputs: Option<proto::MultimodalInputs>, | ||
| tool_call_constraint: Option<(String, String)>, | ||
| ) -> Result<proto::GenerateRequest, String> { | ||
| let sampling_params = | ||
| Self::build_grpc_sampling_params_from_messages(body, tool_call_constraint)?; | ||
|
|
||
| let grpc_request = proto::GenerateRequest { | ||
| request_id, | ||
| tokenized: Some(proto::TokenizedInput { | ||
| original_text: processed_text, | ||
| input_ids: token_ids, | ||
| }), | ||
| mm_inputs: multimodal_inputs, | ||
| sampling_params: Some(sampling_params), | ||
| return_logprob: false, | ||
| logprob_start_len: -1, | ||
| top_logprobs_num: 0, | ||
| return_hidden_states: false, | ||
| stream: body.stream.unwrap_or(false), | ||
| ..Default::default() | ||
| }; | ||
|
|
||
| Ok(grpc_request) | ||
| } | ||
|
|
||
| /// Build gRPC SamplingParams from CreateMessageRequest | ||
| fn build_grpc_sampling_params_from_messages( | ||
| request: &CreateMessageRequest, | ||
| tool_call_constraint: Option<(String, String)>, | ||
| ) -> Result<proto::SamplingParams, String> { | ||
| let stop_sequences = request.stop_sequences.clone().unwrap_or_default(); | ||
|
|
||
| // skip_special_tokens: false when tools are present (same logic as chat) | ||
| let skip_special_tokens = | ||
| tool_call_constraint.is_none() && request.tools.as_ref().is_none_or(|t| t.is_empty()); | ||
|
|
||
| Ok(proto::SamplingParams { | ||
| temperature: request.temperature.unwrap_or(1.0) as f32, | ||
| top_p: request.top_p.unwrap_or(1.0) as f32, | ||
| top_k: request.top_k.map(|v| v as i32).unwrap_or(-1), | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
Useful? React with 👍 / 👎. |
||
| min_p: 0.0, | ||
| frequency_penalty: 0.0, | ||
| presence_penalty: 0.0, | ||
| repetition_penalty: 1.0, | ||
| max_new_tokens: Some(request.max_tokens), | ||
| stop: stop_sequences, | ||
| stop_token_ids: vec![], | ||
| skip_special_tokens, | ||
| spaces_between_special_tokens: true, | ||
| ignore_eos: false, | ||
| no_stop_trim: false, | ||
| n: 1, | ||
| constraint: Self::build_constraint_for_responses(tool_call_constraint)?, | ||
| ..Default::default() | ||
| }) | ||
| } | ||
|
|
||
| fn build_single_constraint_from_plain( | ||
| params: &GenerateSamplingParams, | ||
| ) -> Result<Option<proto::sampling_params::Constraint>, String> { | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1,9 +1,12 @@ | ||
| //! Messages API endpoint pipeline stages | ||
| //! | ||
| //! These stages handle Messages API-specific preprocessing. | ||
| //! Request building and response processing will be added in follow-up PRs. | ||
| //! These stages handle Messages API-specific preprocessing and request building. | ||
| //! Response processing will be added in a follow-up PR. | ||
|
|
||
| mod preparation; | ||
| mod request_building; | ||
|
|
||
| #[expect(unused_imports, reason = "wired in follow-up PR (pipeline factory)")] | ||
| pub(crate) use preparation::MessagePreparationStage; | ||
| #[expect(unused_imports, reason = "wired in follow-up PR (pipeline factory)")] | ||
| pub(crate) use request_building::MessageRequestBuildingStage; |
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
This logic treats any non-empty
request.toolsas a signal to keep special tokens, even when tool use is effectively disabled (for exampletool_choice: none, which yields no tool constraint) or when tools were filtered out earlier for gRPC use. In those cases generation is plain text, butskip_special_tokensis forced tofalse, which can leak backend control/special tokens into user-visible output; the same pattern is also present in the new vLLM Messages builder.Useful? React with 👍 / 👎.