Repository navigation
feat(ci): add GLM-4.6 nightly benchmark coverage #951
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -20,7 +20,7 @@ use crate::validated::Normalizable; | |
| /// This is the main request type for `/v1/messages` endpoint. | ||
| #[serde_with::skip_serializing_none] | ||
| #[derive(Debug, Clone, Serialize, Deserialize, Validate, schemars::JsonSchema)] | ||
| #[validate(schema(function = "validate_mcp_config"))] | ||
| #[validate(schema(function = "validate_message_request"))] | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. These validator changes appear unrelated to GLM-4.6 benchmark coverage. Per CatherineSue's review comment, this PR's scope is adding GLM-4.6 to nightly benchmarks. The Also applies to: 108-155 🤖 Prompt for AI Agents |
||
| pub struct CreateMessageRequest { | ||
| /// The model that will complete your prompt. | ||
| #[validate(length(min = 1, message = "model field is required and cannot be empty"))] | ||
|
|
@@ -105,13 +105,52 @@ impl CreateMessageRequest { | |
| } | ||
| } | ||
|
|
||
| /// Validate that `mcp_servers` is non-empty when `mcp_toolset` tools are present. | ||
| fn validate_mcp_config(req: &CreateMessageRequest) -> Result<(), validator::ValidationError> { | ||
| /// Validate cross-field constraints for Messages API requests. | ||
| fn validate_message_request(req: &CreateMessageRequest) -> Result<(), validator::ValidationError> { | ||
| if req.has_mcp_toolset() && req.mcp_server_configs().is_none() { | ||
| let mut e = validator::ValidationError::new("mcp_servers_required"); | ||
| e.message = Some("mcp_servers is required when mcp_toolset tools are present".into()); | ||
| return Err(e); | ||
| } | ||
|
|
||
| let Some(tool_choice) = &req.tool_choice else { | ||
| return Ok(()); | ||
| }; | ||
|
|
||
| let has_tools = req.tools.as_ref().is_some_and(|tools| !tools.is_empty()); | ||
| let requires_tools = !matches!(tool_choice, ToolChoice::None); | ||
|
|
||
| if requires_tools && !has_tools { | ||
| let mut e = validator::ValidationError::new("tool_choice_requires_tools"); | ||
| e.message = Some( | ||
| "Invalid value for 'tool_choice': 'tool_choice' is only allowed when 'tools' are specified." | ||
| .into(), | ||
| ); | ||
| return Err(e); | ||
| } | ||
|
|
||
| if let ToolChoice::Tool { name, .. } = tool_choice { | ||
| let tool_exists = req.tools.as_ref().is_some_and(|tools| { | ||
| tools.iter().any(|tool| match tool { | ||
| Tool::Custom(tool) => tool.name == *name, | ||
| Tool::ToolSearch(tool) => tool.name == *name, | ||
| Tool::Bash(tool) => tool.name == *name, | ||
| Tool::TextEditor(tool) => tool.name == *name, | ||
| Tool::WebSearch(tool) => tool.name == *name, | ||
| Tool::McpToolset(_) => false, | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
In Useful? React with 👍 / 👎. |
||
| }) | ||
| }); | ||
|
|
||
| if !tool_exists { | ||
| let mut e = validator::ValidationError::new("tool_choice_tool_not_found"); | ||
| e.message = Some( | ||
| format!("Invalid value for 'tool_choice': tool '{name}' not found in 'tools'.") | ||
| .into(), | ||
| ); | ||
| return Err(e); | ||
| } | ||
| } | ||
|
|
||
| Ok(()) | ||
| } | ||
|
|
||
|
|
@@ -1764,6 +1803,46 @@ mod tests { | |
|
|
||
| use super::*; | ||
|
|
||
| fn base_request() -> CreateMessageRequest { | ||
| CreateMessageRequest { | ||
| model: "claude-test".to_string(), | ||
| messages: vec![InputMessage { | ||
| role: Role::User, | ||
| content: InputContent::String("hello".to_string()), | ||
| }], | ||
| max_tokens: 16, | ||
| metadata: None, | ||
| service_tier: None, | ||
| stop_sequences: None, | ||
| stream: None, | ||
| system: None, | ||
| temperature: None, | ||
| thinking: None, | ||
| tool_choice: None, | ||
| tools: None, | ||
| top_k: None, | ||
| top_p: None, | ||
| container: None, | ||
| mcp_servers: None, | ||
| } | ||
| } | ||
|
|
||
| fn custom_tool(name: &str) -> Tool { | ||
| Tool::Custom(CustomTool { | ||
| name: name.to_string(), | ||
| tool_type: None, | ||
| description: Some("test tool".to_string()), | ||
| input_schema: InputSchema { | ||
| schema_type: "object".to_string(), | ||
| properties: None, | ||
| required: None, | ||
| additional: HashMap::new(), | ||
| }, | ||
| defer_loading: None, | ||
| cache_control: None, | ||
| }) | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_tool_mcp_toolset_defer_loading_deserialization() { | ||
| let json = r#"{ | ||
|
|
@@ -1876,6 +1955,69 @@ mod tests { | |
| } | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_tool_choice_auto_requires_tools() { | ||
| let mut request = base_request(); | ||
| request.tool_choice = Some(ToolChoice::Auto { | ||
| disable_parallel_tool_use: None, | ||
| }); | ||
|
|
||
| assert!(request.validate().is_err()); | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_tool_choice_any_requires_tools() { | ||
| let mut request = base_request(); | ||
| request.tool_choice = Some(ToolChoice::Any { | ||
| disable_parallel_tool_use: None, | ||
| }); | ||
|
|
||
| assert!(request.validate().is_err()); | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_tool_choice_specific_tool_requires_tools() { | ||
| let mut request = base_request(); | ||
| request.tool_choice = Some(ToolChoice::Tool { | ||
| name: "get_weather".to_string(), | ||
| disable_parallel_tool_use: None, | ||
| }); | ||
|
|
||
| assert!(request.validate().is_err()); | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_tool_choice_specific_tool_must_exist() { | ||
| let mut request = base_request(); | ||
| request.tool_choice = Some(ToolChoice::Tool { | ||
| name: "get_weather".to_string(), | ||
| disable_parallel_tool_use: None, | ||
| }); | ||
| request.tools = Some(vec![custom_tool("search_web")]); | ||
|
|
||
| assert!(request.validate().is_err()); | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_tool_choice_none_without_tools_is_valid() { | ||
| let mut request = base_request(); | ||
| request.tool_choice = Some(ToolChoice::None); | ||
|
|
||
| assert!(request.validate().is_ok()); | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_tool_choice_specific_tool_is_valid_when_declared() { | ||
| let mut request = base_request(); | ||
| request.tool_choice = Some(ToolChoice::Tool { | ||
| name: "get_weather".to_string(), | ||
| disable_parallel_tool_use: None, | ||
| }); | ||
| request.tools = Some(vec![custom_tool("get_weather")]); | ||
|
|
||
| assert!(request.validate().is_ok()); | ||
| } | ||
|
|
||
| #[test] | ||
| fn test_full_message_with_tool_search_flow_deserialization() { | ||
| // Simulates the full response from Anthropic API with tool search flow | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -102,6 +102,7 @@ def _run_nightly(setup_backend, genai_bench_runner, model_id, worker_count=1, ** | |
| ("Qwen/Qwen3-30B-A3B", "Qwen30b", 4, ["http", "grpc"], {}), | ||
| ("openai/gpt-oss-20b", "GptOss20b", 1, ["http", "grpc"], {}), | ||
| ("minimaxai/minimax-m2", "MinimaxM2", 1, ["http", "grpc"], {}), | ||
| ("zai-org/GLM-4.6", "Glm46", 1, ["http", "grpc"], {}), | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
The generator at Lines 145-151 will create 🤖 Prompt for AI Agents |
||
| ( | ||
| "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", | ||
| "Llama4Maverick", | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Fix YAMLlint's
bracesviolation.This new inline mapping is already being flagged by static analysis, so the workflow will stay red until the spacing matches the repo's YAMLlint rule.
🧹 Minimal fix
📝 Committable suggestion
🧰 Tools
🪛 YAMLlint (1.38.0)
[error] 349-349: too many spaces inside braces
(braces)
[error] 349-349: too many spaces inside braces
(braces)
🤖 Prompt for AI Agents