Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 4 additions & 4 deletions .github/workflows/unit-test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -37,12 +37,12 @@ jobs:
pip install accelerate
pip install sentence_transformers

- name: Test Frontend Language
- name: Test Backend Runtime
run: |
cd test/lang
cd test/srt
python3 run_suite.py --suite minimal

- name: Test Backend Runtime
- name: Test Frontend Language
run: |
cd test/srt
cd test/lang
python3 run_suite.py --suite minimal
24 changes: 12 additions & 12 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -167,17 +167,7 @@ python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct
- If the model does not have a template in the Hugging Face tokenizer, you can specify a [custom chat template](docs/en/custom_chat_template.md).
- To enable fp8 quantization, you can add `--quantization fp8` on a fp16 checkpoint or directly load a fp8 checkpoint without specifying any arguments.
- To enable experimental torch.compile support, you can add `--enable-torch-compile`. It accelerates small models on small batch sizes.

### Use Models From ModelScope
To use model from [ModelScope](https://www.modelscope.cn), setting environment variable SGLANG_USE_MODELSCOPE.
```
export SGLANG_USE_MODELSCOPE=true
```
Launch [Qwen2-7B-Instruct](https://www.modelscope.cn/models/qwen/qwen2-7b-instruct) Server
```
SGLANG_USE_MODELSCOPE=true python -m sglang.launch_server --model-path qwen/Qwen2-7B-Instruct --port 30000
```


### Supported Models

- Llama / Llama 2 / Llama 3 / Llama 3.1
Expand All @@ -203,7 +193,17 @@ SGLANG_USE_MODELSCOPE=true python -m sglang.launch_server --model-path qwen/Qwen

Instructions for supporting a new model are [here](https://github.com/sgl-project/sglang/blob/main/docs/en/model_support.md).

### Run Llama 3.1 405B
#### Use Models From ModelScope
To use model from [ModelScope](https://www.modelscope.cn), setting environment variable SGLANG_USE_MODELSCOPE.
```
export SGLANG_USE_MODELSCOPE=true
```
Launch [Qwen2-7B-Instruct](https://www.modelscope.cn/models/qwen/qwen2-7b-instruct) Server
```
SGLANG_USE_MODELSCOPE=true python -m sglang.launch_server --model-path qwen/Qwen2-7B-Instruct --port 30000
```

#### Run Llama 3.1 405B

```bash
## Run 405B (fp8) on a single node
Expand Down
5 changes: 4 additions & 1 deletion docs/en/contributor_guide.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,9 @@ Use these commands to format your code and pass CI linting tests.
```
pip3 install pre-commit
cd sglang
pre-commit install .
pre-commit install
pre-commit run --all-files
```

## Add Unit Tests
Add unit tests under [sglang/test](../../test). You can learn how to add and run tests from the README.md in that folder.
7 changes: 5 additions & 2 deletions python/sglang/srt/managers/tp_worker.py
Original file line number Diff line number Diff line change
Expand Up @@ -461,8 +461,11 @@ def forward_prefill_batch(self, batch: ScheduleBatch):
next_token_ids = next_token_ids.tolist()
else:
if self.tokenizer is None:
for i, req in enumerate(batch.reqs):
next_token_ids.extend(req.sampling_params.stop_token_ids)
next_token_ids = []
for req in batch.reqs:
next_token_ids.append(
next(iter(req.sampling_params.stop_token_ids))
)
else:
next_token_ids = [self.tokenizer.eos_token_id] * len(batch.reqs)

Expand Down
6 changes: 4 additions & 2 deletions python/sglang/test/test_programs.py
Original file line number Diff line number Diff line change
Expand Up @@ -149,7 +149,7 @@ def decode_json(s):
assert isinstance(js_obj["population"], int)


def test_expert_answer():
def test_expert_answer(check_answer=True):
@sgl.function
def expert_answer(s, question):
s += "Question: " + question + "\n"
Expand All @@ -167,7 +167,9 @@ def expert_answer(s, question):
)

ret = expert_answer.run(question="What is the capital of France?", temperature=0.1)
assert "paris" in ret.text().lower()

if check_answer:
assert "paris" in ret.text().lower(), f"Answer: {ret.text()}"


def test_tool_use():
Expand Down
32 changes: 19 additions & 13 deletions test/README.md
Original file line number Diff line number Diff line change
@@ -1,26 +1,32 @@
# Run Unit Tests

## Test Frontend Language
```
cd sglang/test/lang
export OPENAI_API_KEY=sk-*****
SGLang uses the built-in library [unittest](https://docs.python.org/3/library/unittest.html) as the testing framework.

## Test Backend Runtime
```bash
cd sglang/test/srt

# Run a single file
python3 test_openai_backend.py
python3 test_srt_endpoint.py

# Run a suite
# Run a single test
python3 -m unittest test_srt_endpoint.TestSRTEndpoint.test_simple_decode

# Run a suite with multiple files
python3 run_suite.py --suite minimal
```

## Test Backend Runtime
```
cd sglang/test/srt
## Test Frontend Language
```bash
cd sglang/test/lang
export OPENAI_API_KEY=sk-*****

# Run a single file
python3 test_eval_accuracy.py
python3 test_openai_backend.py

# Run a single test
python3 -m unittest test_openai_backend.TestOpenAIBackend.test_few_shot_qa

# Run a suite
# Run a suite with multiple files
python3 run_suite.py --suite minimal
```


9 changes: 1 addition & 8 deletions test/lang/test_anthropic_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,11 +21,4 @@ def test_stream(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# from sglang.global_config import global_config

# global_config.verbosity = 2
# t = TestAnthropicBackend()
# t.setUpClass()
# t.test_mt_bench()
unittest.main()
6 changes: 1 addition & 5 deletions test/lang/test_bind_cache.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,8 +48,4 @@ def few_shot_qa(s, prompt, question):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# t = TestBind()
# t.setUpClass()
# t.test_cache()
unittest.main()
7 changes: 1 addition & 6 deletions test/lang/test_choices.py
Original file line number Diff line number Diff line change
Expand Up @@ -87,9 +87,4 @@ def test_unconditional_likelihood_normalized(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# t = TestChoices()
# t.test_token_length_normalized()
# t.test_greedy_token_selection()
# t.test_unconditional_likelihood_normalized()
unittest.main()
2 changes: 1 addition & 1 deletion test/lang/test_litellm_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,4 +21,4 @@ def test_stream(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")
unittest.main()
9 changes: 1 addition & 8 deletions test/lang/test_openai_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -88,11 +88,4 @@ def test_chat_completion_speculative(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# from sglang.global_config import global_config

# global_config.verbosity = 2
# t = TestOpenAIBackend()
# t.setUpClass()
# t.test_stream()
unittest.main()
10 changes: 1 addition & 9 deletions test/lang/test_srt_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,12 +61,4 @@ def test_regex(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# from sglang.global_config import global_config

# global_config.verbosity = 2
# t = TestSRTBackend()
# t.setUpClass()
# t.test_few_shot_qa()
# t.tearDownClass()
unittest.main()
5 changes: 1 addition & 4 deletions test/lang/test_tracing.py
Original file line number Diff line number Diff line change
Expand Up @@ -125,7 +125,4 @@ def tip_suggestion(s):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# t = TestTracing()
# t.test_multi_function()
unittest.main()
21 changes: 5 additions & 16 deletions test/lang/test_vertexai_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,26 +14,22 @@

class TestVertexAIBackend(unittest.TestCase):
backend = None
chat_backend = None
chat_vision_backend = None

@classmethod
def setUpClass(cls):
cls.backend = VertexAI("gemini-pro")
cls.chat_backend = VertexAI("gemini-pro")
cls.chat_vision_backend = VertexAI("gemini-pro-vision")
cls.backend = VertexAI("gemini-1.5-pro-001")

def test_few_shot_qa(self):
set_default_backend(self.backend)
test_few_shot_qa()

def test_mt_bench(self):
set_default_backend(self.chat_backend)
set_default_backend(self.backend)
test_mt_bench()

def test_expert_answer(self):
set_default_backend(self.backend)
test_expert_answer()
test_expert_answer(check_answer=False)

def test_parallel_decoding(self):
set_default_backend(self.backend)
Expand All @@ -44,7 +40,7 @@ def test_parallel_encoding(self):
test_parallel_encoding()

def test_image_qa(self):
set_default_backend(self.chat_vision_backend)
set_default_backend(self.backend)
test_image_qa()

def test_stream(self):
Expand All @@ -53,11 +49,4 @@ def test_stream(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# from sglang.global_config import global_config

# global_config.verbosity = 2
# t = TestVertexAIBackend()
# t.setUpClass()
# t.test_stream()
unittest.main()
2 changes: 1 addition & 1 deletion test/srt/run_suite.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,9 +6,9 @@
suites = {
"minimal": [
"test_eval_accuracy.py",
"test_embedding_openai_server.py",
"test_openai_server.py",
"test_vision_openai_server.py",
"test_embedding_openai_server.py",
"test_chunked_prefill.py",
"test_torch_compile.py",
"test_models_from_modelscope.py",
Expand Down
7 changes: 1 addition & 6 deletions test/srt/test_chunked_prefill.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,9 +37,4 @@ def test_mmlu(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# t = TestAccuracy()
# t.setUpClass()
# t.test_mmlu()
# t.tearDownClass()
unittest.main()
16 changes: 4 additions & 12 deletions test/srt/test_embedding_openai_server.py
Original file line number Diff line number Diff line change
@@ -1,11 +1,8 @@
import json
import time
import unittest

import openai

from sglang.srt.hf_transformers_utils import get_tokenizer
from sglang.srt.openai_api.protocol import EmbeddingObject
from sglang.srt.utils import kill_child_process
from sglang.test.test_utils import popen_launch_server

Expand Down Expand Up @@ -65,12 +62,12 @@ def run_embedding(self, use_list_input, token_input):
), f"{response.usage.total_tokens} vs {num_prompt_tokens}"

def run_batch(self):
# FIXME not implemented
# FIXME: not implemented
pass

def test_embedding(self):
# TODO the fields of encoding_format, dimensions, user are skipped
# TODO support use_list_input
# TODO: the fields of encoding_format, dimensions, user are skipped
# TODO: support use_list_input
for use_list_input in [False, True]:
for token_input in [False, True]:
self.run_embedding(use_list_input, token_input)
Expand All @@ -80,9 +77,4 @@ def test_batch(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# t = TestOpenAIServer()
# t.setUpClass()
# t.test_embedding()
# t.tearDownClass()
unittest.main()
7 changes: 1 addition & 6 deletions test/srt/test_eval_accuracy.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,9 +32,4 @@ def test_mmlu(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# t = TestAccuracy()
# t.setUpClass()
# t.test_mmlu()
# t.tearDownClass()
unittest.main()
2 changes: 1 addition & 1 deletion test/srt/test_models_from_modelscope.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,4 +44,4 @@ def test_prepare_tokenizer(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")
unittest.main()
7 changes: 1 addition & 6 deletions test/srt/test_openai_server.py
Original file line number Diff line number Diff line change
Expand Up @@ -399,9 +399,4 @@ def test_regex(self):


if __name__ == "__main__":
unittest.main(warnings="ignore")

# t = TestOpenAIServer()
# t.setUpClass()
# t.test_completion()
# t.tearDownClass()
unittest.main()
Loading