From 28e5e387a2f7eb64a3a35a1bc777c5e3cf6ea4c8 Mon Sep 17 00:00:00 2001 From: shanmugamr1992 Date: Fri, 24 Jul 2026 09:15:41 -0700 Subject: [PATCH 1/4] test: add async scheduling dynamic inference functional test Add gpt_dynamic_inference_tp1_pp1_583m_async_sched, exercising the new --inference-dynamic-batching-async-sched-mode async path end-to-end with greedy sampling and comparing generated tokens against golden values. Co-Authored-By: Claude Opus 4.7 Signed-off-by: shanmugamr1992 Signed-off-by: Lawrence McAfee --- .../golden_values_dev_dgx_h100.json | 711 ++++++++++++++++++ .../model_config.yaml | 59 ++ .../recipes/h100/gpt-dynamic-inference.yaml | 5 + 3 files changed, 775 insertions(+) create mode 100644 tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_h100.json create mode 100644 tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/model_config.yaml diff --git a/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_h100.json b/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_h100.json new file mode 100644 index 00000000000..af2242f77a9 --- /dev/null +++ b/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_h100.json @@ -0,0 +1,711 @@ +{ + "0": { + "input_prompt": "The $500 Cup of coffee?\nConsider this, most Americans spend an average of $1,500-2,000 a year on this bean water.\nI have a few question for you: \nHow has business been the past few months?\nDo you ever feel like your business is stuck?\nDon't feel like you're able to improve performance and make changes required to achieve success ?\nAre your customers spneding less and less and wanting more?\nHave the gas prices affected your business?\nDo you have employees and do they hate you or wish they could quit?\n\nNow, before you and I can decide wheter or not I will be a good fit for your business we should talk this over with coffee.\nAnd, just to warn you this isn't some casual thing. This is not a date or time to be personal or social (but by all means share what you will coz I'll gladly listen).\nTher eare two major talking points and stratagies we will focios on in our lil coffee social\nFor one, we will find your unique selling Proposition (USP).\nDo have the best price? Are you the cheapest in town? Are your customers jerks? Do you haVE REGULARS? Why do people come back?\nwe'll also look for the holes in your business bucket. I'm willing to bet there's a hole or two in your business we'll find together that'll make this 500 dollar cup of Joe pay for itse;f immedietly.\nMany find themselves to be more profitable by just finding out where the dollars are escaping in their business and I like to think of myself as a guy that comes along with some spakel or putty and patch those holes up for you.\nBeleive me, just fixing one hole can mean a lot...just think about a sinking boat that has a hole in it that's about 3\u201d in diameter... it doesn't take long to sink.\nI have no agenda, besides f=getting to know your business and seeing wher I can patch the holes and find what makes you do darn unique (I know this won't take long.)\nMany folks, I bet, will find what they need to get off their chest with a quick phone call and they just paypal me the money and make a coffee at home. Look, that's fine too.\nI just to get you ot of your comfort zone, because this is where it all starts my frind.\nSome smart GOAT entrepreneur will probably get everything they need just out of our lil mini consulatant for the more extensive business consukting I offer, and look, that's fine, too.\nMaybe this cup of coffee will be all you need to gtet back on your feet, but not only surive, but thrive!\nI'm not trying to make frineds, or make a bunch of money, or look down your shirt\nBut this is only going to be a 45 minute (max) coffee chat\nAnd, I know you (and me) both have a lot on our plates. So no messing around\nAfter our consultation I will follow up with you in a few days and see how things are going, then I will be emailing you about once or twice every two weeks, just to make sure you're staying on task and implementing what we went over.\nTghere is no obligation to go any further and will gladly give you back your money if this pre-consultation doesn't put you on the right path or you don't get any value out of it...", + "generated_text": " $ is a$ is a $ is a $ is a $ is a $ is a $$1, you\n$ $$ $\n$ $- $\n$\n$\n$ $$$\n$\n$\n$\n$\n$\n$\n$\n$???????, $\n-You\n$\n$-1\n$\n$\nHow\n$.?\n\n$\n$\n1.1\nI\n$.\nI\n\n\nHow\n$,,,0,1,0,00- $\n... is\nThe\nThe\n", + "generated_tokens": [ + 1659, + 1395, + 1261, + 1036, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1036, + 1049, + 1044, + 1636, + 1010, + 1036, + 1659, + 1036, + 1659, + 1010, + 1036, + 1659, + 1045, + 1659, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1659, + 1036, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1063, + 1063, + 1063, + 1063, + 1063, + 1063, + 1063, + 1044, + 1659, + 1010, + 1045, + 4568, + 1010, + 1036, + 1010, + 1036, + 1045, + 1049, + 1010, + 1036, + 1010, + 1036, + 1010, + 7801, + 1010, + 1036, + 1046, + 9551, + 1010, + 1036, + 1010, + 1036, + 1010, + 1049, + 1046, + 1049, + 1010, + 1073, + 1010, + 1036, + 1046, + 1010, + 1073, + 1010, + 1010, + 1010, + 7801, + 1010, + 1036, + 1044, + 1044, + 1044, + 1048, + 1044, + 1049, + 1044, + 1048, + 1044, + 1048, + 1048, + 1045, + 1659, + 1010, + 2880, + 1395, + 1010, + 1784, + 1010, + 1784, + 1010 + ], + "latency": 10.892494678497314, + "ttft": 0.194596529006958, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "32": { + "input_prompt": "create a conversational article", + "generated_text": " about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article.", + "generated_tokens": [ + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046 + ], + "latency": 10.882038831710815, + "ttft": 0.1947627067565918, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "64": { + "input_prompt": "App Concept: \"Eggy's Interactive Adventure World\"\n\nEggy's Interactive Adventure World is a cutting-edge app designed for phones and tablets that combines immersive storytelling with engaging, interactive experiences for both parents and children. This innovative app allows users to explore the world of Eggy and participate in a series of adventures that teach essential life skills, foster personal growth, and strengthen parent-child bonds.\n\nKey Features:\n\n1. Interactive Stories: The app brings the 20 Eggy stories to life through interactive, animated storytelling. Users can read, listen, and actively participate in the stories, making choices that guide Eggy through various adventures, each with unique lessons and experiences.\n2. Augmented Reality (AR) Integration: The app utilizes AR technology, allowing users to interact with Eggy and the story characters in their real-world environment. This immersive experience creates a deeper connection with the stories and encourages engagement.\n3. Personalized Adventures: The app customizes each story based on the user's interests, needs, and personal growth areas. By collecting user input, the app tailors the storylines and learning outcomes to ensure that each adventure resonates with the child's unique experiences.\n4. Parent-Child Cooperative Play: The app encourages parents and children to collaborate and engage in the stories together. Through a split-screen mode or multi-device connectivity, parents and children can actively participate in the stories and learn essential life skills together.\n5. Skill-Building Mini-Games: Interspersed throughout the stories are educational mini-games that reinforce the lessons and themes. These games are designed to be fun, engaging, and accessible to children, encouraging them to practice and develop various skills.\n6. Progress Tracking and Rewards: The app tracks the user's progress through the stories and rewards them with badges, trophies, and other collectibles. This gamification aspect encourages continued engagement and fosters a sense of accomplishment and growth.\n7. Community Features: The app provides a safe, moderated online community where parents can share their experiences, discuss the stories and lessons, and connect with other families who are also using the app. This community fosters support, inspiration, and learning.\n8. Expert Resources: The app includes access to resources from experts in child development, psychology, and education, providing parents with additional tools and guidance on fostering the growth and well-being of their children.\n\nEggy's Interactive Adventure World is a one-of-a-kind, mind-blowing app that revolutionizes the way parents and children engage with stories and learn essential life skills. Combining the power of storytelling, cutting-edge technology, and expert resources, this app is a must-have for families looking to strengthen their bonds and support their children's personal growth.", + "generated_text": "The 1999\u20132000 season was the 10th season of the National Hockey League (NHL). The 1999\u20132000 season was the 10th season of the National Hockey League (NHL). The 1999\u20132000 season was the 10th season of the National Hockey League (NHL). The 1999\u20132000 season was the 10th season was the 10th season was the 10 season was the 10 season was the 10", + "generated_tokens": [ + 2, + 1784, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1307, + 1278, + 5805, + 46982, + 8525, + 1319, + 19644, + 1076, + 1577, + 1531, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1307, + 1278, + 5805, + 46982, + 8525, + 1319, + 19644, + 1076, + 1577, + 1531, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1307, + 1278, + 5805, + 46982, + 8525, + 1319, + 19644, + 1076, + 1577, + 1531, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048 + ], + "latency": 10.872116804122925, + "ttft": 0.19449090957641602, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "96": { + "input_prompt": "what is the concept of double materiality in sustainability?", + "generated_text": " What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double material", + "generated_tokens": [ + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318 + ], + "latency": 10.860994100570679, + "ttft": 0.18096256256103516, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "127": { + "input_prompt": "as before we talk about using , i usually do if else() to in case the property would be null before.\nif today i public a property and i want to use using to instead , will code be like this?\n\npublic Material Mat;\n\nusing(Mat){\n// Do something;\n}\n\nis that right?", + "generated_text": " i am not sure if i am using the wrong syntax or if i am using the syntax correctly but i am not sure if i am using the syntax correctly or if i am using the syntax incorrectly.\n\nI am not sure if i am using the wrong syntax or if i am using the syntax correctly but i am not sure if i am using the syntax correctly or if i am using the syntax incorrectly.\n\nI am not sure if i am using the wrong syntax or if i am using the syntax correctly but i am not sure if i am using the syntax correctly or if i am using the syntax incorrectly.\n\nI am not sure if i am using", + "generated_tokens": [ + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 8462, + 22692, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1809, + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 73751, + 1338, + 1073, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 8462, + 22692, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1809, + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 73751, + 1338, + 1073, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 8462, + 22692, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1809, + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 73751, + 1338, + 1073, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505 + ], + "latency": 10.852606058120728, + "ttft": 0.17280030250549316, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "throughput": [ + 1130.0929743054994, + 1467.7527907005106, + 1467.8709552178896 + ], + "mem-max-allocated-bytes": 22954507776, + "lifetime_prefill_token_count": 28887, + "async_sched_step_count": 131, + "async_sched_compaction_step_count": 4 +} \ No newline at end of file diff --git a/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/model_config.yaml b/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/model_config.yaml new file mode 100644 index 00000000000..e4c49d32c68 --- /dev/null +++ b/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/model_config.yaml @@ -0,0 +1,59 @@ +ENV_VARS: + CUDA_DEVICE_MAX_CONNECTIONS: 1 + NVTE_ALLOW_NONDETERMINISTIC_ALGO: 0 + NCCL_ALGO: Ring + CUBLAS_WORKSPACE_CONFIG: :4096:8 +TEST_TYPE: frozen-start +MODE: inference +MODEL_ARGS: + --tiktoken-pattern: v2 + --use-mcore-models: true + --tokenizer-type: TikTokenizer + --tokenizer-model: ${CHECKPOINT_LOAD_PATH}/model/mcore_mistral/nemo_minitron-0.5b/v1/multiMixV8.gpt4o_nc_sd.500000.128k.vocab.json + --auto-detect-ckpt-format: true + --max-tokens-to-oom: 3600000 + --inference-max-seq-length: 4096 + --attention-backend: flash + --use-checkpoint-args: true + --micro-batch-size: 1 + --no-load-optim: true + --no-use-tokenizer-model-from-checkpoint-args: true + --timing-log-level: 0 + --load: ${CHECKPOINT_LOAD_PATH}/model/mcore_mistral/nemo_minitron-0.5b/v1 + --distributed-backend: nccl + --log-interval: 1 + --transformer-impl: transformer_engine + --tensor-model-parallel-size: 1 + --pipeline-model-parallel-size: 1 + --deterministic-mode: true + --ckpt-format: torch_dist + --bf16: true + --log-memory-to-tensorboard: true + --log-num-zeros-in-grad: true + --log-validation-ppl-to-tensorboard: true + --log-timers-to-tensorboard: true + --num-layers: 24 + --hidden-size: 1152 + --num-attention-heads: 16 + --max-position-embeddings: 1024 + --seq-length: 1024 + --temperature: 1.0 + --top_k: 1 + # Async scheduling only supports greedy sampling (top_k=1, top_p=0.0) and does + # not support log probabilities, stop words, chunked prefill, or prefix + # caching (see dynamic_engine._validate_async_sched_support_for_request). + --inference-dynamic-batching-buffer-size-gb: 20 + --inference-dynamic-batching-async-sched-mode: async + --dist-ckpt-strictness: log_unexpected + --inference-ckpt-non-strict: true # To handle the extra_state errors + --output-path: ${INFERENCE_OUTPUT_PATH} + --output-every-n-results: 32 + --prompt-file: ${DATA_PATH}/text/sharegpt-vicuna/filtered/processed.jsonl + --prompt-file-num-truncate: 128 # originally 1024 + --num-tokens-to-generate: 128 # originally 512 + --incoming-requests-per-step: 32 + --termination-id: -1 + --inference-repeat-n: 3 + --inference-logging-step-interval: 1 +METRICS: + - "generated_tokens" diff --git a/tests/test_utils/recipes/h100/gpt-dynamic-inference.yaml b/tests/test_utils/recipes/h100/gpt-dynamic-inference.yaml index 43661c16cd3..79e97beb5f0 100644 --- a/tests/test_utils/recipes/h100/gpt-dynamic-inference.yaml +++ b/tests/test_utils/recipes/h100/gpt-dynamic-inference.yaml @@ -178,3 +178,8 @@ products: - environment: [dev] scope: [mr] platforms: [dgx_h100] + - test_case: [gpt_dynamic_inference_tp1_pp1_583m_async_sched] + products: + - environment: [dev] + scope: [mr] + platforms: [dgx_h100] From 1f932f7c4928b88dfbeed19115377c400b7ef673 Mon Sep 17 00:00:00 2001 From: shanmugamr1992 Date: Fri, 24 Jul 2026 10:32:36 -0700 Subject: [PATCH 2/4] test: add GB200 async scheduling dynamic inference functional test Register the gpt_dynamic_inference_tp1_pp1_583m_async_sched test case on dgx_gb200 via a new gb200/gpt-dynamic-inference.yaml recipe (nodes: 1, gpus: 4) and capture platform-specific golden values. Generated tokens match the H100 golden (deterministic greedy sampling); throughput is GB200-specific. Co-Authored-By: Claude Opus 4.7 Signed-off-by: shanmugamr1992 Signed-off-by: Lawrence McAfee --- .../golden_values_dev_dgx_gb200.json | 711 ++++++++++++++++++ .../recipes/gb200/gpt-dynamic-inference.yaml | 65 ++ 2 files changed, 776 insertions(+) create mode 100644 tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_gb200.json create mode 100644 tests/test_utils/recipes/gb200/gpt-dynamic-inference.yaml diff --git a/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_gb200.json b/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_gb200.json new file mode 100644 index 00000000000..5737d10fce9 --- /dev/null +++ b/tests/functional_tests/test_cases/gpt/gpt_dynamic_inference_tp1_pp1_583m_async_sched/golden_values_dev_dgx_gb200.json @@ -0,0 +1,711 @@ +{ + "0": { + "input_prompt": "The $500 Cup of coffee?\nConsider this, most Americans spend an average of $1,500-2,000 a year on this bean water.\nI have a few question for you: \nHow has business been the past few months?\nDo you ever feel like your business is stuck?\nDon't feel like you're able to improve performance and make changes required to achieve success ?\nAre your customers spneding less and less and wanting more?\nHave the gas prices affected your business?\nDo you have employees and do they hate you or wish they could quit?\n\nNow, before you and I can decide wheter or not I will be a good fit for your business we should talk this over with coffee.\nAnd, just to warn you this isn't some casual thing. This is not a date or time to be personal or social (but by all means share what you will coz I'll gladly listen).\nTher eare two major talking points and stratagies we will focios on in our lil coffee social\nFor one, we will find your unique selling Proposition (USP).\nDo have the best price? Are you the cheapest in town? Are your customers jerks? Do you haVE REGULARS? Why do people come back?\nwe'll also look for the holes in your business bucket. I'm willing to bet there's a hole or two in your business we'll find together that'll make this 500 dollar cup of Joe pay for itse;f immedietly.\nMany find themselves to be more profitable by just finding out where the dollars are escaping in their business and I like to think of myself as a guy that comes along with some spakel or putty and patch those holes up for you.\nBeleive me, just fixing one hole can mean a lot...just think about a sinking boat that has a hole in it that's about 3\u201d in diameter... it doesn't take long to sink.\nI have no agenda, besides f=getting to know your business and seeing wher I can patch the holes and find what makes you do darn unique (I know this won't take long.)\nMany folks, I bet, will find what they need to get off their chest with a quick phone call and they just paypal me the money and make a coffee at home. Look, that's fine too.\nI just to get you ot of your comfort zone, because this is where it all starts my frind.\nSome smart GOAT entrepreneur will probably get everything they need just out of our lil mini consulatant for the more extensive business consukting I offer, and look, that's fine, too.\nMaybe this cup of coffee will be all you need to gtet back on your feet, but not only surive, but thrive!\nI'm not trying to make frineds, or make a bunch of money, or look down your shirt\nBut this is only going to be a 45 minute (max) coffee chat\nAnd, I know you (and me) both have a lot on our plates. So no messing around\nAfter our consultation I will follow up with you in a few days and see how things are going, then I will be emailing you about once or twice every two weeks, just to make sure you're staying on task and implementing what we went over.\nTghere is no obligation to go any further and will gladly give you back your money if this pre-consultation doesn't put you on the right path or you don't get any value out of it...", + "generated_text": " $ is a$ is a $ is a $ is a $ is a $ is a $$1, you\n$ $$ $\n$ $- $\n$\n$\n$ $$$\n$\n$\n$\n$\n$\n$\n$\n$???????, $\n-You\n$\n$-1\n$\n$\nHow\n$.?\n\n$\n$\n1.1\nI\n$.\nI\n\n\nHow\n$,,,0,1,0,00- $\n... is\nThe\nThe\n", + "generated_tokens": [ + 1659, + 1395, + 1261, + 1036, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1395, + 1261, + 1659, + 1036, + 1049, + 1044, + 1636, + 1010, + 1036, + 1659, + 1036, + 1659, + 1010, + 1036, + 1659, + 1045, + 1659, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1659, + 1036, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1010, + 1036, + 1063, + 1063, + 1063, + 1063, + 1063, + 1063, + 1063, + 1044, + 1659, + 1010, + 1045, + 4568, + 1010, + 1036, + 1010, + 1036, + 1045, + 1049, + 1010, + 1036, + 1010, + 1036, + 1010, + 7801, + 1010, + 1036, + 1046, + 9551, + 1010, + 1036, + 1010, + 1036, + 1010, + 1049, + 1046, + 1049, + 1010, + 1073, + 1010, + 1036, + 1046, + 1010, + 1073, + 1010, + 1010, + 1010, + 7801, + 1010, + 1036, + 1044, + 1044, + 1044, + 1048, + 1044, + 1049, + 1044, + 1048, + 1044, + 1048, + 1048, + 1045, + 1659, + 1010, + 2880, + 1395, + 1010, + 1784, + 1010, + 1784, + 1010 + ], + "latency": 15.173143148422241, + "ttft": 0.27222132682800293, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "32": { + "input_prompt": "create a conversational article", + "generated_text": " about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article. The article should be about the topic of the article.", + "generated_tokens": [ + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046, + 1531, + 9369, + 2715, + 1402, + 2314, + 1278, + 17915, + 1307, + 1278, + 9369, + 1046 + ], + "latency": 15.158410787582397, + "ttft": 0.2701425552368164, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "64": { + "input_prompt": "App Concept: \"Eggy's Interactive Adventure World\"\n\nEggy's Interactive Adventure World is a cutting-edge app designed for phones and tablets that combines immersive storytelling with engaging, interactive experiences for both parents and children. This innovative app allows users to explore the world of Eggy and participate in a series of adventures that teach essential life skills, foster personal growth, and strengthen parent-child bonds.\n\nKey Features:\n\n1. Interactive Stories: The app brings the 20 Eggy stories to life through interactive, animated storytelling. Users can read, listen, and actively participate in the stories, making choices that guide Eggy through various adventures, each with unique lessons and experiences.\n2. Augmented Reality (AR) Integration: The app utilizes AR technology, allowing users to interact with Eggy and the story characters in their real-world environment. This immersive experience creates a deeper connection with the stories and encourages engagement.\n3. Personalized Adventures: The app customizes each story based on the user's interests, needs, and personal growth areas. By collecting user input, the app tailors the storylines and learning outcomes to ensure that each adventure resonates with the child's unique experiences.\n4. Parent-Child Cooperative Play: The app encourages parents and children to collaborate and engage in the stories together. Through a split-screen mode or multi-device connectivity, parents and children can actively participate in the stories and learn essential life skills together.\n5. Skill-Building Mini-Games: Interspersed throughout the stories are educational mini-games that reinforce the lessons and themes. These games are designed to be fun, engaging, and accessible to children, encouraging them to practice and develop various skills.\n6. Progress Tracking and Rewards: The app tracks the user's progress through the stories and rewards them with badges, trophies, and other collectibles. This gamification aspect encourages continued engagement and fosters a sense of accomplishment and growth.\n7. Community Features: The app provides a safe, moderated online community where parents can share their experiences, discuss the stories and lessons, and connect with other families who are also using the app. This community fosters support, inspiration, and learning.\n8. Expert Resources: The app includes access to resources from experts in child development, psychology, and education, providing parents with additional tools and guidance on fostering the growth and well-being of their children.\n\nEggy's Interactive Adventure World is a one-of-a-kind, mind-blowing app that revolutionizes the way parents and children engage with stories and learn essential life skills. Combining the power of storytelling, cutting-edge technology, and expert resources, this app is a must-have for families looking to strengthen their bonds and support their children's personal growth.", + "generated_text": "The 1999\u20132000 season was the 10th season of the National Hockey League (NHL). The 1999\u20132000 season was the 10th season of the National Hockey League (NHL). The 1999\u20132000 season was the 10th season of the National Hockey League (NHL). The 1999\u20132000 season was the 10th season was the 10th season was the 10 season was the 10 season was the 10", + "generated_tokens": [ + 2, + 1784, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1307, + 1278, + 5805, + 46982, + 8525, + 1319, + 19644, + 1076, + 1577, + 1531, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1307, + 1278, + 5805, + 46982, + 8525, + 1319, + 19644, + 1076, + 1577, + 1531, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1307, + 1278, + 5805, + 46982, + 8525, + 1319, + 19644, + 1076, + 1577, + 1531, + 1032, + 1049, + 1057, + 1057, + 1057, + 1882, + 1050, + 1048, + 1048, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 1411, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048, + 5526, + 1486, + 1278, + 1032, + 1049, + 1048 + ], + "latency": 15.145662069320679, + "ttft": 0.26699352264404297, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "96": { + "input_prompt": "what is the concept of double materiality in sustainability?", + "generated_text": " What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double materiality in sustainability? What is the concept of double material", + "generated_tokens": [ + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318, + 1532, + 1294, + 60187, + 1063, + 5675, + 1395, + 1278, + 7401, + 1307, + 5558, + 4318 + ], + "latency": 15.136452198028564, + "ttft": 0.24875521659851074, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "127": { + "input_prompt": "as before we talk about using , i usually do if else() to in case the property would be null before.\nif today i public a property and i want to use using to instead , will code be like this?\n\npublic Material Mat;\n\nusing(Mat){\n// Do something;\n}\n\nis that right?", + "generated_text": " i am not sure if i am using the wrong syntax or if i am using the syntax correctly but i am not sure if i am using the syntax correctly or if i am using the syntax incorrectly.\n\nI am not sure if i am using the wrong syntax or if i am using the syntax correctly but i am not sure if i am using the syntax correctly or if i am using the syntax incorrectly.\n\nI am not sure if i am using the wrong syntax or if i am using the syntax correctly but i am not sure if i am using the syntax correctly or if i am using the syntax incorrectly.\n\nI am not sure if i am using", + "generated_tokens": [ + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 8462, + 22692, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1809, + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 73751, + 1338, + 1073, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 8462, + 22692, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1809, + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 73751, + 1338, + 1073, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 8462, + 22692, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1809, + 1623, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 17047, + 1505, + 1693, + 1623, + 1855, + 2505, + 1278, + 22692, + 73751, + 1338, + 1073, + 1855, + 1605, + 5257, + 1693, + 1623, + 1855, + 2505 + ], + "latency": 15.125433683395386, + "ttft": 0.23802447319030762, + "cuda_graph_request_count_map": null, + "step_count": 132, + "top_n_logprobs": null, + "prompt_top_n_logprobs": null + }, + "throughput": [ + 836.2814759906043, + 1044.265060439428, + 1053.2105377765834 + ], + "mem-max-allocated-bytes": 22954507776, + "lifetime_prefill_token_count": 28887, + "async_sched_step_count": 131, + "async_sched_compaction_step_count": 4 +} \ No newline at end of file diff --git a/tests/test_utils/recipes/gb200/gpt-dynamic-inference.yaml b/tests/test_utils/recipes/gb200/gpt-dynamic-inference.yaml new file mode 100644 index 00000000000..6a81fab3b9a --- /dev/null +++ b/tests/test_utils/recipes/gb200/gpt-dynamic-inference.yaml @@ -0,0 +1,65 @@ +type: basic +format_version: 1 +maintainers: [mcore] +loggers: [stdout] +spec: + name: '{test_case}_{environment}_{platforms}' + model: gpt + build: mcore-pyt-{environment} + nodes: 1 + gpus: 4 + n_repeat: 1 + platforms: dgx_gb200 + script_setup: | + set -euo pipefail + unset https_proxy + echo "machine gitlab-master.nvidia.com login okoenig password $RO_API_TOKEN" | tee -a /root/.netrc + + # Checkout latest + cd /opt + rm -rf /opt/megatron-lm; mkdir megatron-lm; cd megatron-lm + git init + git remote add origin $MCORE_REPO + git fetch origin '+refs/merge-requests/*:refs/remotes/merge-requests/*' + git fetch origin $MCORE_MR_COMMIT + git checkout $MCORE_MR_COMMIT + git rev-parse HEAD + # Checkout backwards-ref + cd /opt + rm -rf /opt/megatron-lm-legacy; mkdir megatron-lm-legacy; cd megatron-lm-legacy + git init + git remote add origin $MCORE_REPO + git fetch origin $MCORE_BACKWARDS_COMMIT + git checkout $MCORE_BACKWARDS_COMMIT + git rev-parse HEAD + rm -rf megatron; cp -a /opt/megatron-lm/megatron ./ + script: |- + set -euo pipefail + ls + cd /opt/megatron-lm + export GPUS_PER_NODE={gpus} + + ARGUMENTS=( + "CHECKPOINT_LOAD_PATH=/mnt/artifacts" + "CHECKPOINT_SAVE_PATH=/tmp/checkpoints" + "DATA_PATH=/mnt/artifacts" + "DATA_CACHE_PATH=/workspace/data/cache" + "TRAINING_SCRIPT_PATH=examples/inference/advanced/gpt_dynamic_inference.py" + "TRAINING_PARAMS_PATH=./tests/functional_tests/test_cases/{model}/{test_case}/model_config.yaml" + "GOLDEN_VALUES_PATH=./tests/functional_tests/test_cases/{model}/{test_case}/golden_values_{environment}_{platforms}.json" + "OUTPUT_PATH={assets_dir}" + "TENSORBOARD_PATH={assets_dir}/tensorboard" + "INFERENCE_OUTPUT_PATH={assets_dir}/golden_values_{environment}_{platforms}.json" + "N_REPEAT={n_repeat}" + "ENABLE_LIGHTWEIGHT_MODE=${{ENABLE_LIGHTWEIGHT_MODE:-}}" + "RECORD_CHECKPOINTS=${{RECORD_CHECKPOINTS:-}}" + ) + + bash ./tests/functional_tests/shell_test_utils/run_ci_test.sh ${{ARGUMENTS[@]}} + +products: + - test_case: [gpt_dynamic_inference_tp1_pp1_583m_async_sched] + products: + - environment: [dev] + scope: [mr] + platforms: [dgx_gb200] From ea31595e9d4e6e9d1547add6cd4ea9e906bee39e Mon Sep 17 00:00:00 2001 From: shanmugamr1992 Date: Fri, 24 Jul 2026 18:02:09 -0700 Subject: [PATCH 3/4] test: add async scheduling inference perf tests (H100 + GB200) Adds inference performance tests exercising dynamic-batching async prefill scheduling on both H100 (DP=8) and GB200 (DP=4) for the 583M checkpoint. - run_perf_test.sh: read ASYNC_SCHED config key; when true, launch the server with --inference-dynamic-batching-async-sched-mode async. Also pass --skip-prompt-log-probs because the server tool force-sets return_log_probs=True (echo/loglikelihood support), which would set materialize_only_last_token_logits to False and violate the async-scheduling requirement. - gpt_583m_perf_async_sched (H100) and gpt_583m_perf_async_sched_gb200_4gpu (GB200) test cases with platform-keyed baselines captured on cw-dfw / oci-hsg. - Register both in the existing h100/gpt-perf-dp8 and gb200/gpt-perf-dp4 recipes. Co-Authored-By: Claude Opus 4.7 Signed-off-by: shanmugamr1992 Signed-off-by: Lawrence McAfee --- .../shell_test_utils/run_perf_test.sh | 20 +++++++ .../baseline_values.json | 52 +++++++++++++++++++ .../model_config.yaml | 38 ++++++++++++++ .../baseline_values.json | 40 ++++++++++++++ .../model_config.yaml | 36 +++++++++++++ .../recipes/gb200/gpt-perf-dp4.yaml | 2 +- .../test_utils/recipes/h100/gpt-perf-dp8.yaml | 2 +- 7 files changed, 188 insertions(+), 2 deletions(-) create mode 100644 tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/baseline_values.json create mode 100644 tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/model_config.yaml create mode 100644 tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json create mode 100644 tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/model_config.yaml diff --git a/tests/performance_tests/shell_test_utils/run_perf_test.sh b/tests/performance_tests/shell_test_utils/run_perf_test.sh index f39196cda98..830af93c5e0 100755 --- a/tests/performance_tests/shell_test_utils/run_perf_test.sh +++ b/tests/performance_tests/shell_test_utils/run_perf_test.sh @@ -88,6 +88,12 @@ NUM_TIMED_ITERS=$("$YQ" '.NUM_TIMED_ITERS // 5' "$CONFIG_PATH") # hybrid models should use 'gsm8k' — synthetic input gives misleading # perf because every token is identical (uniform expert routing, hot KV). DATASET=$("$YQ" '.DATASET // "synthetic"' "$CONFIG_PATH") +# Async prefill scheduling for dynamic batching. When true, the server is +# launched with --inference-dynamic-batching-async-sched-mode async (overlaps +# the prefill scheduler with GPU compute). Requires greedy sampling / no +# logprobs / no stop words — the static benchmark client already satisfies +# these (temperature 0.0, ignore_eos, no stop tokens, no logprobs requested). +ASYNC_SCHED=$("$YQ" '.ASYNC_SCHED // false' "$CONFIG_PATH") mapfile -t BATCH_SIZES < <("$YQ" '.BATCH_SIZES[]' "$CONFIG_PATH") # For MoE configs, expert-parallelism is orthogonal to DP and reshapes the @@ -194,6 +200,20 @@ SERVER_COMMON_ARGS=( --host 0.0.0.0 ) +# Enable async prefill scheduling when the test case opts in. Async scheduling +# requires materialize_only_last_token_logits=True. run_dynamic_text_generation_server +# force-sets return_log_probs=True (for echo/loglikelihood support), which would +# flip materialize_only_last_token_logits to False; passing --skip-prompt-log-probs +# keeps it True (materialize = not(return_log_probs and not skip_prompt_log_probs)). +# The perf client never requests prompt logprobs, so this is a no-op for the metrics. +if [[ "$ASYNC_SCHED" == "true" ]]; then + echo "[run_perf_test] async scheduling enabled (--inference-dynamic-batching-async-sched-mode async --skip-prompt-log-probs)" + SERVER_COMMON_ARGS+=( + --inference-dynamic-batching-async-sched-mode async + --skip-prompt-log-probs + ) +fi + ( cd "$ROOT_DIR" uv run --no-sync python -m torch.distributed.run \ diff --git a/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/baseline_values.json b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/baseline_values.json new file mode 100644 index 00000000000..026a7354660 --- /dev/null +++ b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/baseline_values.json @@ -0,0 +1,52 @@ +{ + "h100": { + "batch_1": { + "batch_size": 1, + "dataset": "synthetic", + "num_input_tokens_avg": 512.0, + "num_output_tokens": 128, + "num_iters": 5, + "throughput_tok_per_sec": 42.75601962213067, + "avg_latency_ms": 2993.6806965619326, + "p50_latency_ms": 2992.9271759465337, + "p99_latency_ms": 3003.761636093259, + "tpot_ms_per_tok": 23.388519530999474 + }, + "batch_8": { + "batch_size": 8, + "dataset": "synthetic", + "num_input_tokens_avg": 512.0, + "num_output_tokens": 128, + "num_iters": 5, + "throughput_tok_per_sec": 344.0237043263471, + "avg_latency_ms": 2920.4657254274935, + "p50_latency_ms": 2921.021580696106, + "p99_latency_ms": 2982.457813806832, + "tpot_ms_per_tok": 23.254211554012727 + }, + "batch_32": { + "batch_size": 32, + "dataset": "synthetic", + "num_input_tokens_avg": 512.0, + "num_output_tokens": 128, + "num_iters": 5, + "throughput_tok_per_sec": 1372.033198644746, + "avg_latency_ms": 2927.9737344244495, + "p50_latency_ms": 2925.4276445135474, + "p99_latency_ms": 3030.2867460995913, + "tpot_ms_per_tok": 23.32305080635706 + }, + "batch_128": { + "batch_size": 128, + "dataset": "synthetic", + "num_input_tokens_avg": 512.0, + "num_output_tokens": 128, + "num_iters": 5, + "throughput_tok_per_sec": 5395.836458577838, + "avg_latency_ms": 2958.10645565507, + "p50_latency_ms": 2953.0187863856554, + "p99_latency_ms": 3037.2949857264757, + "tpot_ms_per_tok": 23.72199398232624 + } + } +} diff --git a/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/model_config.yaml b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/model_config.yaml new file mode 100644 index 00000000000..d95f385ebe4 --- /dev/null +++ b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched/model_config.yaml @@ -0,0 +1,38 @@ +# Inference perf test: 583M mcore-mistral checkpoint, TP=1 PP=1 DP=8 (8 GPUs), +# async prefill scheduling enabled. +# +# Same 583M checkpoint / DP=8 dynamic-batching path as gpt_583m_perf, but the +# server runs with --inference-dynamic-batching-async-sched-mode async (the +# scheduler overlaps prefill setup with GPU compute). Measures the throughput / +# latency of the async scheduling path exercised by the functional test +# gpt_dynamic_inference_tp1_pp1_583m_async_sched. +# +# Async scheduling requires greedy sampling / no logprobs / no stop words; the +# static benchmark client already satisfies these (temperature 0.0, ignore_eos). +# +# Baseline values are recorded by `RECORD_BASELINE=1 run_perf_test.sh ...` +# and compared on subsequent runs with TOLERANCE_PCT tolerance. + +MODEL: gpt_583m +TP: 1 +PP: 1 +DP: 8 +ASYNC_SCHED: true +NUM_INPUT_TOKENS: 512 +NUM_OUTPUT_TOKENS: 128 +NUM_WARMUP_ITERS: 2 +NUM_TIMED_ITERS: 5 +BATCH_SIZES: + - 1 + - 8 + - 32 + - 128 +TOLERANCE_PCT: 10 +# p99 omitted on purpose: with NUM_TIMED_ITERS=5 it is the max of 5 samples, +# not a real percentile, so it produces flaky regressions even when throughput +# / avg / p50 are stable. p99 is still recorded in results.json for visibility. +METRICS: + - throughput_tok_per_sec + - avg_latency_ms + - p50_latency_ms + - tpot_ms_per_tok diff --git a/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json new file mode 100644 index 00000000000..f946a9fe815 --- /dev/null +++ b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json @@ -0,0 +1,40 @@ +{ + "gb200": { + "batch_8": { + "batch_size": 8, + "dataset": "synthetic", + "num_input_tokens_avg": 512.0, + "num_output_tokens": 128, + "num_iters": 5, + "throughput_tok_per_sec": 297.1264485236512, + "avg_latency_ms": 3367.1278487032396, + "p50_latency_ms": 3376.3205889845267, + "p99_latency_ms": 3464.05855397461, + "tpot_ms_per_tok": 26.924563732882234 + }, + "batch_32": { + "batch_size": 32, + "dataset": "synthetic", + "num_input_tokens_avg": 512.0, + "num_output_tokens": 128, + "num_iters": 5, + "throughput_tok_per_sec": 1002.4383866367954, + "avg_latency_ms": 3545.93873367412, + "p50_latency_ms": 3405.740808986593, + "p99_latency_ms": 5780.234997044317, + "tpot_ms_per_tok": 31.92216142815596 + }, + "batch_128": { + "batch_size": 128, + "dataset": "synthetic", + "num_input_tokens_avg": 512.0, + "num_output_tokens": 128, + "num_iters": 5, + "throughput_tok_per_sec": 4696.658921234499, + "avg_latency_ms": 3424.1397720949863, + "p50_latency_ms": 3426.1854289798066, + "p99_latency_ms": 3504.938129975926, + "tpot_ms_per_tok": 27.253416129769903 + } + } +} diff --git a/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/model_config.yaml b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/model_config.yaml new file mode 100644 index 00000000000..16bc84ab5c5 --- /dev/null +++ b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/model_config.yaml @@ -0,0 +1,36 @@ +# Inference perf test: 583M mcore-mistral checkpoint, TP=1 PP=1 DP=4 (4 GPUs), +# async prefill scheduling enabled. +# +# GB200 single-node variant of gpt_583m_perf_async_sched. GB200 nodes have a +# 4-GPU/node limit, so the DP=8 configuration used on H100 doesn't fit on a +# single GB200 node. This is a separate test (different world size, different +# baseline) — not a multi-node port of the DP=8 case. +# +# The server runs with --inference-dynamic-batching-async-sched-mode async. +# Async scheduling requires greedy sampling / no logprobs / no stop words; the +# static benchmark client already satisfies these (temperature 0.0, ignore_eos). + +MODEL: gpt_583m +TP: 1 +PP: 1 +DP: 4 +ASYNC_SCHED: true +NUM_INPUT_TOKENS: 512 +NUM_OUTPUT_TOKENS: 128 +# 2 warmup iters can leave the first timed iteration cold on GB200, poisoning +# the mean/tail metrics; warm up more so timing starts at steady state. +NUM_WARMUP_ITERS: 5 +NUM_TIMED_ITERS: 5 +# batch_size=1 is omitted: at single-stream the 4-GPU (DP=4) deployment is +# barely utilized, so per-iteration jitter dominates and the small-sample p50 +# latency is too noisy to gate on. Larger batches are stable perf signals. +BATCH_SIZES: + - 8 + - 32 + - 128 +TOLERANCE_PCT: 10 +METRICS: + - throughput_tok_per_sec + - avg_latency_ms + - p50_latency_ms + - tpot_ms_per_tok diff --git a/tests/test_utils/recipes/gb200/gpt-perf-dp4.yaml b/tests/test_utils/recipes/gb200/gpt-perf-dp4.yaml index 9f4839d28fe..71a9dccf1b2 100644 --- a/tests/test_utils/recipes/gb200/gpt-perf-dp4.yaml +++ b/tests/test_utils/recipes/gb200/gpt-perf-dp4.yaml @@ -40,7 +40,7 @@ spec: GPUS_PER_NODE=4 bash ./tests/performance_tests/shell_test_utils/run_perf_test.sh ${{ARGUMENTS[@]}} products: - - test_case: [gpt_583m_perf_gb200_4gpu] + - test_case: [gpt_583m_perf_gb200_4gpu, gpt_583m_perf_async_sched_gb200_4gpu] products: - environment: [dev] scope: [mr] diff --git a/tests/test_utils/recipes/h100/gpt-perf-dp8.yaml b/tests/test_utils/recipes/h100/gpt-perf-dp8.yaml index 6d9cc4e948f..7484989358f 100644 --- a/tests/test_utils/recipes/h100/gpt-perf-dp8.yaml +++ b/tests/test_utils/recipes/h100/gpt-perf-dp8.yaml @@ -38,7 +38,7 @@ spec: GPUS_PER_NODE=8 bash ./tests/performance_tests/shell_test_utils/run_perf_test.sh ${{ARGUMENTS[@]}} products: - - test_case: [gpt_583m_perf] + - test_case: [gpt_583m_perf, gpt_583m_perf_async_sched] products: - environment: [dev] scope: [mr] From b15d0405c17244fca5a6c9fce2f84f91eb489efc Mon Sep 17 00:00:00 2001 From: shanmugamr1992 Date: Mon, 27 Jul 2026 12:02:43 -0700 Subject: [PATCH 4/4] test: re-record GB200 async-sched baseline (fix batch-32 outlier) The initial GB200 batch-32 capture (1002 tok/s, TPOT 31.9) was a noisy cold outlier; a fresh capture puts it at ~1217 tok/s (TPOT 26.3), in line with the non-async DP=4 baseline. Re-recording avoids a too-low floor that would push normal runs near the +20% improvement ceiling. Co-Authored-By: Claude Opus 4.7 Signed-off-by: shanmugamr1992 Signed-off-by: Lawrence McAfee --- .../baseline_values.json | 30 +++++++++---------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json index f946a9fe815..122e3399183 100644 --- a/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json +++ b/tests/performance_tests/test_cases/gpt/gpt_583m_perf_async_sched_gb200_4gpu/baseline_values.json @@ -6,11 +6,11 @@ "num_input_tokens_avg": 512.0, "num_output_tokens": 128, "num_iters": 5, - "throughput_tok_per_sec": 297.1264485236512, - "avg_latency_ms": 3367.1278487032396, - "p50_latency_ms": 3376.3205889845267, - "p99_latency_ms": 3464.05855397461, - "tpot_ms_per_tok": 26.924563732882234 + "throughput_tok_per_sec": 306.77571598122546, + "avg_latency_ms": 3292.265848722309, + "p50_latency_ms": 3284.7761889570393, + "p99_latency_ms": 3369.791687990073, + "tpot_ms_per_tok": 26.07768341249539 }, "batch_32": { "batch_size": 32, @@ -18,11 +18,11 @@ "num_input_tokens_avg": 512.0, "num_output_tokens": 128, "num_iters": 5, - "throughput_tok_per_sec": 1002.4383866367954, - "avg_latency_ms": 3545.93873367412, - "p50_latency_ms": 3405.740808986593, - "p99_latency_ms": 5780.234997044317, - "tpot_ms_per_tok": 31.92216142815596 + "throughput_tok_per_sec": 1217.121447486444, + "avg_latency_ms": 3322.6687765843963, + "p50_latency_ms": 3319.4968919851817, + "p99_latency_ms": 3425.9593110182323, + "tpot_ms_per_tok": 26.29154228288826 }, "batch_128": { "batch_size": 128, @@ -30,11 +30,11 @@ "num_input_tokens_avg": 512.0, "num_output_tokens": 128, "num_iters": 5, - "throughput_tok_per_sec": 4696.658921234499, - "avg_latency_ms": 3424.1397720949863, - "p50_latency_ms": 3426.1854289798066, - "p99_latency_ms": 3504.938129975926, - "tpot_ms_per_tok": 27.253416129769903 + "throughput_tok_per_sec": 4573.446310856698, + "avg_latency_ms": 3390.523879592547, + "p50_latency_ms": 3371.0599309997633, + "p99_latency_ms": 3771.179543051403, + "tpot_ms_per_tok": 27.987646798464993 } } }