Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions python/sglang/test/test_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -183,6 +183,22 @@ def download_image_with_retry(image_url: str, max_retries: int = 3) -> Image.Ima
time.sleep(2**i)


def build_vlm_image_prompt(processor, question: str) -> str:
# Take the image placeholder from the model's own HF chat template: a
# hand-written one silently degrades to a text-only prompt on any model
# whose placeholder differs.
return processor.apply_chat_template(
[
{
"role": "user",
"content": [{"type": "image"}, {"type": "text", "text": question}],
}
],
tokenize=False,
add_generation_prompt=True,
)


def is_in_ci():
"""Return whether it is in CI runner."""
return get_bool_env_var("SGLANG_IS_IN_CI")
Expand Down
37 changes: 25 additions & 12 deletions test/manual/distributed/test_dp_attention_large.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@

import requests

from sglang.lang.chat_template import get_chat_template_by_model_path
from sglang.srt.utils import kill_process_tree
from sglang.test.kits.ebnf_constrained_kit import EBNFConstrainedMixin
from sglang.test.kits.json_constrained_kit import JSONConstrainedMixin
Expand Down Expand Up @@ -164,24 +163,38 @@ def tearDownClass(cls):
kill_process_tree(cls.process.pid)

def test_vlm_generate(self):
chat_template = get_chat_template_by_model_path(self.model)
prompt = f"{chat_template.image_token}What is in this image?"
# Go through /v1/chat/completions so the server inserts the model's own
# image placeholder instead of the test guessing one.
response = requests.post(
self.base_url + "/generate",
self.base_url + "/v1/chat/completions",
json={
"text": prompt,
"image_data": [self.image_url],
"sampling_params": {
"temperature": 0,
"max_new_tokens": 16,
},
"model": "default",
"messages": [
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {"url": self.image_url},
},
{"type": "text", "text": "What is in this image?"},
],
}
],
"temperature": 0,
"max_tokens": 16,
},
)
response.raise_for_status()
response_json = response.json()
print(response_json)
self.assertIn("output_ids", response_json)
self.assertGreater(len(response_json["output_ids"]), 0)
self.assertTrue(response_json["choices"][0]["message"]["content"])

# image_tokens comes from the prefill's multimodal item offsets, so a
# non-zero count is what proves the image reached the vision tower.
usage_details = response_json["usage"].get("prompt_tokens_details")
self.assertIsNotNone(usage_details, "prompt carried no multimodal tokens")
self.assertGreater(usage_details.get("image_tokens", 0), 0)


if __name__ == "__main__":
Expand Down
8 changes: 5 additions & 3 deletions test/manual/quant/test_torchao.py
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
import unittest

import requests
from transformers import AutoProcessor

from sglang import Engine
from sglang.lang.chat_template import get_chat_template_by_model_path
from sglang.srt.utils import kill_process_tree
from sglang.test.kits.eval_accuracy_kit import MMLUMixin
from sglang.test.test_utils import (
Expand All @@ -13,6 +13,7 @@
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
build_vlm_image_prompt,
is_in_amd_ci,
popen_launch_server,
)
Expand Down Expand Up @@ -72,8 +73,9 @@ def test_throughput(self):
class TestTorchAOForVLM(CustomTestCase):
def test_vlm_generate(self):
model_path = DEFAULT_SMALL_VLM_MODEL_NAME_FOR_TEST
chat_template = get_chat_template_by_model_path(model_path)
text = f"{chat_template.image_token}What is in this picture? Answer: "
text = build_vlm_image_prompt(
AutoProcessor.from_pretrained(model_path), "What is in this picture?"
)

engine = Engine(
model_path=model_path,
Expand Down
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
import unittest

import torch
from transformers import AutoProcessor

from sglang import Engine
from sglang.lang.chat_template import get_chat_template_by_model_path
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.run_eval import run_eval
Expand All @@ -13,6 +13,7 @@
DEFAULT_URL_FOR_TEST,
CustomTestCase,
SimpleNamespace,
build_vlm_image_prompt,
is_in_amd_ci,
popen_launch_server,
)
Expand Down Expand Up @@ -73,8 +74,9 @@ class TestPiecewiseCudaGraphQwen25VLEmbedding(CustomTestCase):

def test_embedding(self):
model_path = "Qwen/Qwen2.5-VL-3B-Instruct"
chat_template = get_chat_template_by_model_path(model_path)
text = f"{chat_template.image_token}What is in this picture? Answer: "
text = build_vlm_image_prompt(
AutoProcessor.from_pretrained(model_path), "What is in this picture?"
)
extra_args = (
{"mem_fraction_static": AMD_MEM_FRACTION_STATIC} if is_in_amd_ci() else {}
)
Expand Down
37 changes: 25 additions & 12 deletions test/registered/dp_attn/test_dp_attention.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,6 @@

import requests

from sglang.lang.chat_template import get_chat_template_by_model_path
from sglang.srt.environ import envs
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
Expand Down Expand Up @@ -204,24 +203,38 @@ def tearDownClass(cls):
kill_process_tree(cls.process.pid)

def test_vlm_generate(self):
chat_template = get_chat_template_by_model_path(self.model)
prompt = f"{chat_template.image_token}What is in this image?"
# Go through /v1/chat/completions so the server inserts the model's own
# image placeholder instead of the test guessing one.
response = requests.post(
self.base_url + "/generate",
self.base_url + "/v1/chat/completions",
json={
"text": prompt,
"image_data": [self.image_url],
"sampling_params": {
"temperature": 0,
"max_new_tokens": 16,
},
"model": "default",
"messages": [
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {"url": self.image_url},
},
{"type": "text", "text": "What is in this image?"},
],
}
],
"temperature": 0,
"max_tokens": 16,
},
)
response.raise_for_status()
response_json = response.json()
print(response_json)
self.assertIn("output_ids", response_json)
self.assertGreater(len(response_json["output_ids"]), 0)
self.assertTrue(response_json["choices"][0]["message"]["content"])

# image_tokens comes from the prefill's multimodal item offsets, so a
# non-zero count is what proves the image reached the vision tower.
usage_details = response_json["usage"].get("prompt_tokens_details")
self.assertIsNotNone(usage_details, "prompt carried no multimodal tokens")
self.assertGreater(usage_details.get("image_tokens", 0), 0)


if __name__ == "__main__":
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,6 @@

import requests

from sglang.lang.chat_template import get_chat_template_by_model_path
from sglang.srt.environ import envs
from sglang.srt.utils import kill_process_tree
from sglang.test.ascend.npu_eval_accuracy_kit import NPUGSM8KMixin
Expand Down Expand Up @@ -175,24 +174,38 @@ def tearDownClass(cls):
kill_process_tree(cls.process.pid)

def test_vlm_generate(self):
chat_template = get_chat_template_by_model_path(self.model)
prompt = f"{chat_template.image_token}What is in this image?"
# Go through /v1/chat/completions so the server inserts the model's own
# image placeholder instead of the test guessing one.
response = requests.post(
self.base_url + "/generate",
self.base_url + "/v1/chat/completions",
json={
"text": prompt,
"image_data": [self.image_url],
"sampling_params": {
"temperature": 0,
"max_new_tokens": 16,
},
"model": "default",
"messages": [
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {"url": self.image_url},
},
{"type": "text", "text": "What is in this image?"},
],
}
],
"temperature": 0,
"max_tokens": 16,
},
)
response.raise_for_status()
response_json = response.json()
print(response_json)
self.assertIn("output_ids", response_json)
self.assertGreater(len(response_json["output_ids"]), 0)
self.assertTrue(response_json["choices"][0]["message"]["content"])

# image_tokens comes from the prefill's multimodal item offsets, so a
# non-zero count is what proves the image reached the vision tower.
usage_details = response_json["usage"].get("prompt_tokens_details")
self.assertIsNotNone(usage_details, "prompt carried no multimodal tokens")
self.assertGreater(usage_details.get("image_tokens", 0), 0)


if __name__ == "__main__":
Expand Down
15 changes: 9 additions & 6 deletions test/registered/tokenizer/test_skip_tokenizer_init.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,6 @@
import requests
from transformers import AutoProcessor, AutoTokenizer

from sglang.lang.chat_template import get_chat_template_by_model_path
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import (
Expand All @@ -19,6 +18,7 @@
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
build_vlm_image_prompt,
download_image_with_retry,
popen_launch_server,
)
Expand Down Expand Up @@ -91,9 +91,13 @@ def assert_one_item(item):
self.assertEqual(item["meta_info"]["prompt_tokens"], len(input_ids))

if return_logprob:
num_input_logprobs = len(input_ids) - request["logprob_start_len"]
if num_input_logprobs > len(input_ids):
num_input_logprobs -= len(input_ids)
# -1 resolves to the prompt end, so no input logprob is returned.
if request["logprob_start_len"] == -1:
num_input_logprobs = 0
else:
num_input_logprobs = (
len(input_ids) - request["logprob_start_len"]
)
self.assertEqual(
len(item["meta_info"]["input_token_logprobs"]),
num_input_logprobs,
Expand Down Expand Up @@ -230,8 +234,7 @@ def setUpClass(cls):
cls.eos_token_id = [cls.tokenizer.eos_token_id]

def get_input_ids(self, _prompt_text) -> list[int]:
chat_template = get_chat_template_by_model_path(self.model)
text = f"{chat_template.image_token}What is in this picture?"
text = build_vlm_image_prompt(self.processor, "What is in this picture?")
inputs = self.processor(
text=[text],
images=[self.image],
Expand Down
Loading