[CI] Migrate nightly tests to test/registered/ (#15582)

This commit is contained in:
Alison Shao
2025-12-22 22:16:32 -08:00
committed by GitHub
parent bc3ca30023
commit 989d4b3012
64 changed files with 168 additions and 140 deletions
@@ -1,15 +1,9 @@
import sys
import unittest
from pathlib import Path
# Add nightly directory to path for run_combined_tests import
sys.path.insert(0, str(Path(__file__).parent.parent.parent / "nightly"))
from accuracy_test_runner import AccuracyTestParams
from performance_test_runner import PerformanceTestParams
from run_combined_tests import run_combined_tests
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
# Runs on both H200 and B200 via nightly-8-gpu-common suite
@@ -1,15 +1,9 @@
import sys
import unittest
from pathlib import Path
# Add nightly directory to path for run_combined_tests import
sys.path.insert(0, str(Path(__file__).parent.parent.parent / "nightly"))
from accuracy_test_runner import AccuracyTestParams
from performance_test_runner import PerformanceTestParams
from run_combined_tests import run_combined_tests
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
register_cuda_ci(est_time=8000, suite="nightly-8-gpu-common", nightly=True)
+3 -9
View File
@@ -1,15 +1,9 @@
import sys
import unittest
from pathlib import Path
# Add nightly directory to path for run_combined_tests import
sys.path.insert(0, str(Path(__file__).parent.parent.parent / "nightly"))
from accuracy_test_runner import AccuracyTestParams
from performance_test_runner import PerformanceTestParams
from run_combined_tests import run_combined_tests
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings
# Runs on both H200 and B200 via nightly-8-gpu-common suite
+3 -9
View File
@@ -1,15 +1,9 @@
import sys
import unittest
from pathlib import Path
# Add nightly directory to path for run_combined_tests import
sys.path.insert(0, str(Path(__file__).parent.parent.parent / "nightly"))
from accuracy_test_runner import AccuracyTestParams
from performance_test_runner import PerformanceTestParams
from run_combined_tests import run_combined_tests
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings
# Runs on both H200 and B200 via nightly-8-gpu-common suite
@@ -1,15 +1,9 @@
import sys
import unittest
from pathlib import Path
# Add nightly directory to path for run_combined_tests import
sys.path.insert(0, str(Path(__file__).parent.parent.parent / "nightly"))
from accuracy_test_runner import AccuracyTestParams
from performance_test_runner import PerformanceTestParams
from run_combined_tests import run_combined_tests
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings
# Runs on both H200 and B200 via nightly-8-gpu-common suite
@@ -1,16 +1,10 @@
import os
import sys
import unittest
from pathlib import Path
# Add nightly directory to path for run_combined_tests import
sys.path.insert(0, str(Path(__file__).parent.parent.parent / "nightly"))
from accuracy_test_runner import AccuracyTestParams
from performance_test_runner import PerformanceTestParams
from run_combined_tests import run_combined_tests
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
# Runs on both H200 and B200 via nightly-8-gpu-common suite
@@ -1,15 +1,9 @@
import sys
import unittest
from pathlib import Path
# Add nightly directory to path for run_combined_tests import
sys.path.insert(0, str(Path(__file__).parent.parent.parent / "nightly"))
from accuracy_test_runner import AccuracyTestParams
from performance_test_runner import PerformanceTestParams
from run_combined_tests import run_combined_tests
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
# Runs on both H200 and B200 via nightly-8-gpu-common suite
@@ -0,0 +1,108 @@
import multiprocessing as mp
import unittest
from typing import Optional
import torch
from transformers import AutoConfig, AutoTokenizer
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.runners import DEFAULT_PROMPTS, HFRunner, SRTRunner
from sglang.test.test_utils import CustomTestCase, get_similarities
register_npu_ci(
est_time=400,
suite="nightly-1-npu-a3",
nightly=True,
disabled="embeddings are not all close",
)
MODELS = [
("/root/.cache/modelscope/hub/models/iic/gte_Qwen2-1.5B-instruct", 1, 1e-5),
("/root/.cache/modelscope/hub/models/Qwen/Qwen3-Embedding-8B", 1, 1e-5),
]
TORCH_DTYPES = [torch.bfloat16]
class TestEmbeddingModels(CustomTestCase):
@classmethod
def setUpClass(cls):
mp.set_start_method("spawn", force=True)
def _truncate_prompts(self, prompts, model_path):
config = AutoConfig.from_pretrained(model_path)
max_length = getattr(config, "max_position_embeddings", 2048)
tokenizer = AutoTokenizer.from_pretrained(model_path)
truncated_prompts = []
for prompt in prompts:
tokens = tokenizer(prompt, return_tensors="pt", truncation=False)
if len(tokens.input_ids[0]) > max_length:
truncated_text = tokenizer.decode(
tokens.input_ids[0][: max_length - 1], skip_special_tokens=True
)
truncated_prompts.append(truncated_text)
else:
truncated_prompts.append(prompt)
return truncated_prompts
def assert_close_prefill_logits(
self,
prompts,
model_path,
tp_size,
torch_dtype,
prefill_tolerance,
matryoshka_dim: Optional[int] = None,
) -> None:
truncated_prompts = self._truncate_prompts(prompts, model_path)
with HFRunner(
model_path,
torch_dtype=torch_dtype,
model_type="embedding",
matryoshka_dim=matryoshka_dim,
) as hf_runner:
hf_outputs = hf_runner.forward(truncated_prompts)
attention_backend = "ascend"
with SRTRunner(
model_path,
tp_size=tp_size,
torch_dtype=torch_dtype,
model_type="embedding",
attention_backend=attention_backend,
json_model_override_args=(
{"matryoshka_dimensions": [matryoshka_dim]} if matryoshka_dim else None
),
) as srt_runner:
srt_outputs = srt_runner.forward(
truncated_prompts, dimensions=matryoshka_dim
)
for i in range(len(prompts)):
hf_logits = torch.Tensor(hf_outputs.embed_logits[i])
srt_logits = torch.Tensor(srt_outputs.embed_logits[i])
similarity = torch.tensor(get_similarities(hf_logits, srt_logits))
print("similarity diff", abs(similarity - 1))
if len(prompts[i]) <= 1000:
assert torch.all(
abs(similarity - 1) < prefill_tolerance
), "embeddings are not all close"
def test_prefill_logits(self):
models_to_test = MODELS
for model, tp_size, prefill_tolerance in models_to_test:
for torch_dtype in TORCH_DTYPES:
self.assert_close_prefill_logits(
DEFAULT_PROMPTS, model, tp_size, torch_dtype, prefill_tolerance
)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/arcee-ai/AFM-4.5B-Base"
accuracy = 0.00
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,21 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(
est_time=400,
suite="nightly-1-npu-a3",
nightly=True,
disabled="The accuracy test result is 0.",
)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/baichuan-inc/Baichuan2-13B-Chat"
accuracy = 0.00
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,91 @@
import os
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.few_shot_gsm8k import run_eval
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
register_npu_ci(
est_time=400,
suite="nightly-2-npu-a3",
nightly=True,
disabled="The accuracy test result is 0.",
)
class TestC4AI(CustomTestCase):
model = "/root/.cache/modelscope/hub/models/CohereForAI/c4ai-command-r-v01"
accuracy = 0.05
@classmethod
def setUpClass(cls):
cls.base_url = DEFAULT_URL_FOR_TEST
chat_template_path = "/__w/sglang/sglang/test/nightly/ascend/llm_models/tool_chat_template_c4ai_command_r_v01.jinja"
other_args = [
"--trust-remote-code",
"--mem-fraction-static",
"0.8",
"--attention-backend",
"ascend",
"--disable-cuda-graph",
"--chat-template",
chat_template_path,
"--tp-size",
"2",
"--dtype",
"bfloat16",
]
env = os.environ.copy()
env.update(
{
"PYTORCH_NPU_ALLOC_CONF": "expandable_segments:True",
"ASCEND_MF_STORE_URL": "tcp://127.0.0.1:24666",
"HCCL_BUFFSIZE": "200",
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK": "24",
"USE_VLLM_CUSTOM_ALLREDUCE": "1",
"HCCL_EXEC_TIMEOUT": "200",
"STREAMS_PER_DEVICE": "32",
"SGLANG_ENABLE_TORCH_COMPILE": "1",
}
)
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
other_args=other_args,
env=env,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_gsm8k(self):
args = SimpleNamespace(
num_shots=5,
data_path=None,
num_questions=200,
max_new_tokens=512,
parallel=128,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval(args)
self.assertGreater(
metrics["accuracy"],
self.accuracy,
f'Accuracy of {self.model} is {str(metrics["accuracy"])}, is lower than {self.accuracy}',
)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,26 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/ZhipuAI/chatglm2-6b"
accuracy = 0.25
other_args = [
"--trust-remote-code",
"--mem-fraction-static",
"0.8",
"--attention-backend",
"ascend",
"--disable-cuda-graph",
"--dtype",
"bfloat16",
]
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,26 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/LGAI-EXAONE/EXAONE-3.5-7.8B-Instruct"
accuracy = 0.00
other_args = [
"--trust-remote-code",
"--mem-fraction-static",
"0.8",
"--attention-backend",
"ascend",
"--disable-cuda-graph",
"--dtype",
"bfloat16",
]
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,21 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(
est_time=400,
suite="nightly-1-npu-a3",
nightly=True,
disabled="The accuracy test result is 0.",
)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/LLM-Research/gemma-3-1b-it"
accuracy = 0.00
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestGLM49BChat(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/ZhipuAI/glm-4-9b-chat"
accuracy = 0.00
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = (
"/root/.cache/modelscope/hub/models/ibm-granite/granite-3.0-3b-a800m-instruct"
)
accuracy = 0.00
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/ibm-granite/granite-3.1-8b-instruct"
accuracy = 0.695
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/Shanghai_AI_Laboratory/internlm2-7b"
accuracy = 0.6
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/inclusionAI/Ling-lite"
accuracy = 0.75
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/LLM-Research/Llama-2-7B"
accuracy = 0.18
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/XiaomiMiMo/MiMo-7B-RL"
accuracy = 0.75
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/mistralai/Mistral-7B-Instruct-v0.2"
accuracy = 0.375
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/Howeee/persimmon-8b-chat"
accuracy = 0.17
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,16 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/LLM-Research/Phi-4-multimodal-instruct"
accuracy = 0.8
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,26 @@
import unittest
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.test_utils import CustomTestCase
register_npu_ci(est_time=400, suite="nightly-1-npu-a3", nightly=True)
class TestMistral7B(GSM8KAscendMixin, CustomTestCase):
model = "/root/.cache/modelscope/hub/models/HuggingFaceTB/SmolLM-1.7B"
accuracy = 0.05
other_args = [
"--trust-remote-code",
"--mem-fraction-static",
"0.8",
"--attention-backend",
"ascend",
"--disable-cuda-graph",
"--dtype",
"bfloat16",
]
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1 @@
{{ bos_token }}{% if messages[0]['role'] == 'system' %}{% set loop_messages = messages[1:] %}{% set system_message = messages[0]['content'] %}{% elif false == true %}{% set loop_messages = messages %}{% set system_message = 'You are Command-R, a brilliant, sophisticated, AI-assistant trained to assist human users by providing thorough responses. You are trained by Cohere.' %}{% else %}{% set loop_messages = messages %}{% set system_message = false %}{% endif %}{% if system_message != false %}{{ '<|START_OF_TURN_TOKEN|><|SYSTEM_TOKEN|>' + system_message + '<|END_OF_TURN_TOKEN|>' }}{% endif %}{% for message in loop_messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{{ '<|START_OF_TURN_TOKEN|><|USER_TOKEN|>' + content.strip() + '<|END_OF_TURN_TOKEN|>' }}{% elif message['role'] == 'assistant' %}{{ '<|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|>' + content.strip() + '<|END_OF_TURN_TOKEN|>' }}{% endif %}{% endfor %}{% if add_generation_prompt %}{{ '<|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|>' }}{% endif %}
@@ -0,0 +1,92 @@
import multiprocessing as mp
import unittest
import torch
from sglang.test.ci.ci_register import register_npu_ci
from sglang.test.runners import TEST_RERANK_QUERY_DOCS, HFRunner, SRTRunner
from sglang.test.test_utils import CustomTestCase
register_npu_ci(
est_time=400,
suite="nightly-1-npu-a3",
nightly=True,
disabled="cross encoder scores are not all close",
)
MODELS = [
("/root/.cache/modelscope/hub/models/BAAI/bge-reranker-v2-m3", 1, 1e-2),
]
ATTENTION_BACKEND = ["ascend"]
TORCH_DTYPES = [torch.bfloat16]
class TestCrossEncoderModels(CustomTestCase):
@classmethod
def setUpClass(cls):
mp.set_start_method("spawn", force=True)
def assert_close_prefill_logits(
self,
prompts,
model_path,
tp_size,
torch_dtype,
score_tolerance,
attention_backend,
) -> None:
with HFRunner(
model_path,
torch_dtype=torch_dtype,
model_type="cross_encoder",
) as hf_runner:
hf_scores = hf_runner.forward(prompts).scores
with SRTRunner(
model_path,
tp_size=tp_size,
torch_dtype=torch_dtype,
model_type="cross_encoder",
attention_backend=attention_backend,
chunked_prefill_size=-1,
disable_radix_cache=True,
) as srt_runner:
srt_scores = srt_runner.forward(prompts).scores
for i in range(len(srt_scores)):
score_difference = abs(hf_scores[i] - srt_scores[i])
assert (
score_difference < score_tolerance
), "cross encoder scores are not all close"
def preprocess_prompts(self, prompt):
processed_prompts = []
query = prompt["query"]
documents = prompt["documents"]
for document in documents:
processed_prompts.append([query, document])
return processed_prompts
def test_prefill_logits(self):
models_to_test = MODELS
for model, tp_size, prefill_tolerance in models_to_test:
for attention_backend in ATTENTION_BACKEND:
for queryDocs in TEST_RERANK_QUERY_DOCS:
prompts = self.preprocess_prompts(queryDocs)
for torch_dtype in TORCH_DTYPES:
self.assert_close_prefill_logits(
prompts,
model,
tp_size,
torch_dtype,
prefill_tolerance,
attention_backend,
)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1 @@
dataset_path: /root/.cache/huggingface/hub/datasets--lmms-lab--MMMU/snapshots/364f2e2eb107b36e07ff4c5a15f5947a759cef47
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestGemmaModels(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/google/gemma-3-4b-it"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestGemmaModels(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/deepseek-ai/Janus-Pro-1B"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestJanusPro7B(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/deepseek-ai/Janus-Pro-7B"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestGemmaModels(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/XiaomiMiMo/MiMo-VL-7B-RL"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestGemmaModels(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/openbmb/MiniCPM-o-2_6"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestGemmaModels(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/openbmb/MiniCPM-V-2_6"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestGemmaModels(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/microsoft/Phi-4-multimodal-instruct"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,18 @@
import unittest
from sglang.test.ascend.vlm_utils import TestVLMModels
from sglang.test.ci.ci_register import register_npu_ci
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
class TestGemmaModels(TestVLMModels):
model = "/root/.cache/modelscope/hub/models/Qwen/Qwen2.5-VL-3B-Instruct"
mmmu_accuracy = 0.2
def test_vlm_mmmu_benchmark(self):
self._run_vlm_mmmu_test()
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,90 @@
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
from sglang.test.test_utils import (
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
try_cached_model,
)
register_cuda_ci(est_time=3600, suite="nightly-8-gpu-b200", nightly=True)
FULL_DEEPSEEK_V3_MODEL_PATH = "deepseek-ai/DeepSeek-V3-0324"
SERVER_LAUNCH_TIMEOUT = 1000
class TestDeepseekR1Fp8Flashinfer(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = try_cached_model(FULL_DEEPSEEK_V3_MODEL_PATH)
cls.base_url = DEFAULT_URL_FOR_TEST
other_args = [
"--trust-remote-code",
"--disable-radix-cache",
"--max-running-requests",
"512",
"--chunked-prefill-size",
"8192",
"--mem-fraction-static",
"0.9",
"--cuda-graph-max-bs",
"128",
"--max-prefill-tokens",
"8192",
"--kv-cache-dtype",
"fp8_e4m3",
"--quantization",
"fp8",
"--tensor-parallel-size",
"8",
"--data-parallel-size",
"1",
"--expert-parallel-size",
"1",
"--scheduler-recv-interval",
"10",
"--stream-interval",
"10",
"--attention-backend",
"trtllm_mla",
"--fp8-gemm-backend",
"flashinfer_trtllm",
"--moe-runner-backend",
"flashinfer_trtllm",
"--enable-symm-mem",
"--model-loader-extra-config",
'{"enable_multithread_load": true,"num_threads": 64}',
]
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=SERVER_LAUNCH_TIMEOUT,
other_args=other_args,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_gsm8k(self):
args = SimpleNamespace(
num_shots=5,
data_path=None,
num_questions=512,
parallel=512,
max_new_tokens=512,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval_few_shot_gsm8k(args)
print(f"Eval accuracy of GSM8K: {metrics=}")
self.assertGreater(metrics["accuracy"], 0.92)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,75 @@
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
from sglang.test.test_utils import (
DEFAULT_URL_FOR_TEST,
CustomTestCase,
is_in_ci,
popen_launch_server,
write_github_step_summary,
)
register_cuda_ci(est_time=900, suite="nightly-4-gpu-b200", nightly=True)
FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4"
SERVER_LAUNCH_TIMEOUT = 1000
class TestDeepseekV3FP4CutlassMoE(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = FULL_DEEPSEEK_V3_FP4_MODEL_PATH
cls.base_url = DEFAULT_URL_FOR_TEST
other_args = [
"--tp",
"4",
"--ep",
"4",
"--attention-backend",
"trtllm_mla",
"--moe-runner-backend",
"flashinfer_cutlass",
"--quantization",
"modelopt_fp4",
"--model-loader-extra-config",
'{"enable_multithread_load": true}',
]
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=SERVER_LAUNCH_TIMEOUT,
other_args=other_args,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_a_gsm8k(
self,
): # Append an "a" to make this test run first (alphabetically) to warm up the server
args = SimpleNamespace(
num_shots=8,
data_path=None,
num_questions=1319,
parallel=1319,
max_new_tokens=512,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval_few_shot_gsm8k(args)
print(f"{metrics=}")
if is_in_ci():
write_github_step_summary(
f"### test_gsm8k (deepseek-v3-fp4-cutlass-moe)\n"
f'{metrics["accuracy"]=:.3f}\n'
)
self.assertGreater(metrics["accuracy"], 0.935)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,65 @@
import os
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
register_cuda_ci(est_time=300, suite="nightly-4-gpu-b200", nightly=True)
class TestFlashinferTrtllmGenAttnBackend(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = "Qwen/Qwen3-Next-80B-A3B-Instruct"
cls.base_url = DEFAULT_URL_FOR_TEST
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
env={**os.environ, "SGLANG_ENABLE_JIT_DEEPGEMM": "False"},
other_args=[
"--attention-backend",
"trtllm_mha",
"--cuda-graph-max-bs",
"512",
"--tp-size",
"4",
"--ep-size",
"4",
"--mem-fraction-static",
"0.7",
"--mamba-ssm-dtype",
"bfloat16",
"--disable-radix-cache",
],
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_gsm8k(self):
args = SimpleNamespace(
num_shots=5,
data_path=None,
num_questions=200,
max_new_tokens=512,
parallel=128,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval(args)
print(f"{metrics=}")
self.assertGreater(metrics["accuracy"], 0.93)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,110 @@
import os
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
register_cuda_ci(est_time=300, suite="nightly-4-gpu-b200", nightly=True)
class TestFlashinferTrtllmGenMoeBackendFP8(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = "Qwen/Qwen3-Next-80B-A3B-Instruct-FP8"
cls.base_url = DEFAULT_URL_FOR_TEST
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
env={**os.environ, "SGLANG_ENABLE_JIT_DEEPGEMM": "False"},
other_args=[
"--attention-backend",
"triton",
"--moe-runner-backend",
"flashinfer_trtllm",
"--tp-size",
"4",
"--ep-size",
"4",
"--mem-fraction-static",
"0.7",
"--mamba-ssm-dtype",
"bfloat16",
],
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_gsm8k(self):
args = SimpleNamespace(
num_shots=5,
data_path=None,
num_questions=200,
max_new_tokens=512,
parallel=128,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval(args)
print(f"{metrics=}")
self.assertGreater(metrics["accuracy"], 0.93)
class TestFlashinferTrtllmGenMoeBackendBF16(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = "Qwen/Qwen3-Next-80B-A3B-Instruct"
cls.base_url = DEFAULT_URL_FOR_TEST
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
other_args=[
"--attention-backend",
"triton",
"--moe-runner-backend",
"flashinfer_trtllm",
"--cuda-graph-max-bs",
"512",
"--tp-size",
"4",
"--ep-size",
"4",
"--mem-fraction-static",
"0.7",
"--mamba-ssm-dtype",
"bfloat16",
],
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_gsm8k(self):
args = SimpleNamespace(
num_shots=5,
data_path=None,
num_questions=200,
max_new_tokens=512,
parallel=128,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval(args)
print(f"{metrics=}")
self.assertGreater(metrics["accuracy"], 0.93)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,68 @@
import unittest
from types import SimpleNamespace
from sglang.srt.utils import get_device_sm, kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
# modelopt_fp4 requires SM 100+ (Blackwell)
register_cuda_ci(est_time=300, suite="nightly-1-gpu", nightly=True)
@unittest.skipIf(
get_device_sm() < 100, "Test requires CUDA SM 100 or higher (Blackwell)"
)
class TestFlashinferTrtllmGenMoeBackend(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = "nvidia/Qwen3-30B-A3B-NVFP4"
cls.base_url = DEFAULT_URL_FOR_TEST
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
other_args=[
"--moe-runner-backend",
"flashinfer_trtllm",
"--quantization",
"modelopt_fp4",
"--trust-remote-code",
"--disable-radix-cache",
"--max-running-requests",
"1024",
"--chunked-prefill-size",
"16384",
"--mem-fraction-static",
"0.89",
"--max-prefill-tokens",
"16384",
],
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_gsm8k(self):
args = SimpleNamespace(
num_shots=8,
data_path=None,
num_questions=1319,
max_new_tokens=512,
parallel=1319,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval(args)
print(f"{metrics=}")
self.assertGreater(metrics["accuracy"], 0.88)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,261 @@
# Adapted from https://github.com/thinking-machines-lab/batch_invariant_ops/blob/main/test_batch_invariance.py
import math
import unittest
import torch
from sglang.srt.batch_invariant_ops import batch_invariant_ops
from sglang.test.ci.ci_register import register_cuda_ci
register_cuda_ci(est_time=10, suite="nightly-1-gpu", nightly=True)
from sglang.srt.batch_invariant_ops.batch_invariant_ops import set_batch_invariant_mode
from sglang.test.test_utils import CustomTestCase
device_type = getattr(torch.accelerator.current_accelerator(), "type", "cpu")
torch.set_default_device(device_type)
# Just to get the logging out of the way
with set_batch_invariant_mode(True):
pass
class TestBatchInvariantOps(CustomTestCase):
@classmethod
def setUpClass(cls):
batch_invariant_ops._ENABLE_MM_COMPARISON_TEST = True
@classmethod
def tearDownClass(cls):
batch_invariant_ops._ENABLE_MM_COMPARISON_TEST = False
def _test_batch_invariance(self, M, K, N, dtype):
"""
Test that matrix operations produce identical results for:
- Method 1: Matrix-vector multiplication (batch size 1)
- Method 2: Matrix-matrix multiplication, then slice (full batch)
"""
a = torch.linspace(-100, 100, M * K, dtype=dtype).reshape(M, K)
# Create non-contiguous tensor
b = torch.linspace(-100, 100, K * N, dtype=dtype).reshape(N, K)
b = b.transpose(0, 1)
# Method 1: Matrix-vector multiplication (batch size 1)
out1 = torch.mm(a[:1], b)
# Method 2: Matrix-matrix multiplication, then slice (full batch)
out2_pre = torch.mm(a, b)
out2 = out2_pre[:1]
# Check if results are identical
diff = (out1 - out2).abs().max()
return diff.item()
def _run_multiple_iterations(self, iters, M, K, N, dtype):
"""Run multiple iterations and collect diff statistics"""
difflist = []
for _ in range(iters):
diff = self._test_batch_invariance(M, K, N, dtype)
difflist.append(diff)
return difflist
def _assert_batch_invariant_results(self, difflist, dtype, test_name):
"""
Assert that in batch-invariant mode:
1. All diffs must not be NaN
2. All diffs must be exactly 0
3. Max, min, and diff of diffs must all be 0
"""
max_diff = max(difflist)
min_diff = min(difflist)
diff_range = max_diff - min_diff
# Check for NaN values
self.assertFalse(
math.isnan(max_diff), f"{test_name}: max_diff is NaN for {dtype}"
)
self.assertFalse(
math.isnan(min_diff), f"{test_name}: min_diff is NaN for {dtype}"
)
self.assertFalse(
math.isnan(diff_range), f"{test_name}: diff_range is NaN for {dtype}"
)
# Check that all diffs are exactly 0
self.assertEqual(
max_diff,
0.0,
f"{test_name}: max_diff must be 0 in batch-invariant mode, got {max_diff} for {dtype}",
)
self.assertEqual(
min_diff,
0.0,
f"{test_name}: min_diff must be 0 in batch-invariant mode, got {min_diff} for {dtype}",
)
self.assertEqual(
diff_range,
0.0,
f"{test_name}: diff_range must be 0 in batch-invariant mode, got {diff_range} for {dtype}",
)
def test_small_matrices(self):
"""Test batch invariance with small matrix sizes"""
test_cases = [
("Small-1", 8, 64, 128),
("Small-2", 16, 128, 256),
("Small-3", 4, 32, 64),
]
for name, M, K, N in test_cases:
with self.subTest(name=name, M=M, K=K, N=N):
for dtype in [torch.float32, torch.bfloat16]:
with self.subTest(dtype=dtype):
# Run with batch-invariant mode
with set_batch_invariant_mode(True):
difflist = self._run_multiple_iterations(
iters=5, M=M, K=K, N=N, dtype=dtype
)
self._assert_batch_invariant_results(difflist, dtype, name)
def test_medium_matrices(self):
"""Test batch invariance with medium matrix sizes"""
test_cases = [
("Medium-1", 32, 128, 1024),
("Medium-2", 64, 512, 2048),
("Medium-3", 24, 192, 768),
]
for name, M, K, N in test_cases:
with self.subTest(name=name, M=M, K=K, N=N):
for dtype in [torch.float32, torch.bfloat16]:
with self.subTest(dtype=dtype):
# Run with batch-invariant mode
with set_batch_invariant_mode(True):
difflist = self._run_multiple_iterations(
iters=5, M=M, K=K, N=N, dtype=dtype
)
self._assert_batch_invariant_results(difflist, dtype, name)
def test_large_matrices(self):
"""Test batch invariance with large matrix sizes"""
test_cases = [
("Large-1", 128, 1024, 4096),
("Large-2", 256, 2048, 8192),
("Large-3", 96, 768, 3072),
]
for name, M, K, N in test_cases:
with self.subTest(name=name, M=M, K=K, N=N):
for dtype in [torch.float32, torch.bfloat16]:
with self.subTest(dtype=dtype):
# Run with batch-invariant mode
with set_batch_invariant_mode(True):
difflist = self._run_multiple_iterations(
iters=5, M=M, K=K, N=N, dtype=dtype
)
self._assert_batch_invariant_results(difflist, dtype, name)
def test_without_batch_invariant_mode(self):
"""
Test that without batch-invariant mode, results may differ.
This test demonstrates the difference batch-invariant mode makes.
"""
M, K, N = 32, 128, 1024
dtype = torch.float32
# Run without batch-invariant mode
with set_batch_invariant_mode(False):
difflist = self._run_multiple_iterations(
iters=5, M=M, K=K, N=N, dtype=dtype
)
print(f"Without batch-invariant mode, we get diffs: {difflist}")
def _test_bmm_batch_invariance(self, B, M, K, N, dtype):
"""
Test that BMM operations produce identical results for:
- Method 1: BMM with subset of batches
- Method 2: BMM with all batches, then slice
"""
a = torch.linspace(-100, 100, B * M * K, dtype=dtype).reshape(B, M, K)
b = torch.linspace(-100, 100, B * K * N, dtype=dtype).reshape(B, K, N)
# Method 1: BMM with subset (first 2 batches)
subset_size = min(2, B)
out1 = torch.bmm(a[:subset_size], b[:subset_size])
# Method 2: BMM with all batches, then slice
out2_pre = torch.bmm(a, b)
out2 = out2_pre[:subset_size]
# Check if results are identical
diff = (out1 - out2).abs().max()
return diff.item()
def _run_bmm_multiple_iterations(self, iters, B, M, K, N, dtype):
"""Run multiple BMM iterations and collect diff statistics"""
difflist = []
for _ in range(iters):
diff = self._test_bmm_batch_invariance(B, M, K, N, dtype)
difflist.append(diff)
return difflist
def test_bmm_small_matrices(self):
"""Test BMM batch invariance with small matrix sizes"""
test_cases = [
("BMM-Small-1", 4, 8, 64, 128),
("BMM-Small-2", 8, 16, 128, 256),
("BMM-Small-3", 6, 4, 32, 64),
]
for name, B, M, K, N in test_cases:
with self.subTest(name=name, B=B, M=M, K=K, N=N):
for dtype in [torch.float32, torch.bfloat16]:
with self.subTest(dtype=dtype):
# Run with batch-invariant mode
with set_batch_invariant_mode(True):
difflist = self._run_bmm_multiple_iterations(
iters=5, B=B, M=M, K=K, N=N, dtype=dtype
)
self._assert_batch_invariant_results(difflist, dtype, name)
def test_bmm_medium_matrices(self):
"""Test BMM batch invariance with medium matrix sizes"""
test_cases = [
("BMM-Medium-1", 8, 32, 128, 1024),
("BMM-Medium-2", 16, 64, 512, 2048),
("BMM-Medium-3", 12, 24, 192, 768),
]
for name, B, M, K, N in test_cases:
with self.subTest(name=name, B=B, M=M, K=K, N=N):
for dtype in [torch.float32, torch.bfloat16]:
with self.subTest(dtype=dtype):
# Run with batch-invariant mode
with set_batch_invariant_mode(True):
difflist = self._run_bmm_multiple_iterations(
iters=5, B=B, M=M, K=K, N=N, dtype=dtype
)
self._assert_batch_invariant_results(difflist, dtype, name)
def test_bmm_large_matrices(self):
"""Test BMM batch invariance with large matrix sizes"""
test_cases = [
("BMM-Large-1", 16, 128, 1024, 4096),
("BMM-Large-2", 32, 256, 2048, 8192),
("BMM-Large-3", 24, 96, 768, 3072),
]
for name, B, M, K, N in test_cases:
with self.subTest(name=name, B=B, M=M, K=K, N=N):
for dtype in [torch.float32, torch.bfloat16]:
with self.subTest(dtype=dtype):
# Run with batch-invariant mode
with set_batch_invariant_mode(True):
difflist = self._run_bmm_multiple_iterations(
iters=5, B=B, M=M, K=K, N=N, dtype=dtype
)
self._assert_batch_invariant_results(difflist, dtype, name)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,50 @@
import unittest
from types import SimpleNamespace
from sglang.srt.environ import envs
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.run_eval import run_eval
from sglang.test.test_utils import (
DEFAULT_MODEL_NAME_FOR_TEST,
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
register_cuda_ci(est_time=60, suite="nightly-1-gpu", nightly=True)
class TestCppRadixCache(CustomTestCase):
@classmethod
def setUpClass(cls):
envs.SGLANG_EXPERIMENTAL_CPP_RADIX_TREE.set(True)
cls.model = DEFAULT_MODEL_NAME_FOR_TEST
cls.base_url = DEFAULT_URL_FOR_TEST
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_mmlu(self):
args = SimpleNamespace(
base_url=self.base_url,
model=self.model,
eval_name="mmlu",
num_examples=64,
num_threads=32,
)
metrics = run_eval(args)
print(metrics)
self.assertGreaterEqual(metrics["score"], 0.65)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,57 @@
"""
Usage:
cd test/srt
python3 -m unittest test_deepseek_v3_deterministic.TestFa3Deterministic
"""
import unittest
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.test_deterministic_utils import (
COMMON_SERVER_ARGS,
TestDeterministicBase,
)
register_cuda_ci(est_time=240, suite="nightly-1-gpu", nightly=True)
DEEPSEEK_MODEL = "lmsys/sglang-ci-dsv3-test"
class TestFa3Deterministic(TestDeterministicBase):
@classmethod
def get_model(cls):
return DEEPSEEK_MODEL
# Test with fa3 attention backend
@classmethod
def get_server_args(cls):
args = COMMON_SERVER_ARGS
args.extend(
[
"--attention-backend",
"fa3",
]
)
return args
class TestTritonDeterministic(TestDeterministicBase):
@classmethod
def get_model(cls):
return DEEPSEEK_MODEL
# Test with triton attention backend
@classmethod
def get_server_args(cls):
args = COMMON_SERVER_ARGS
args.extend(
[
"--attention-backend",
"triton",
]
)
return args
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,47 @@
"""
Usage:
cd test/srt
python3 -m unittest test_qwen3_next_deterministic.TestFlashInferDeterministic
"""
import unittest
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.test_deterministic_utils import (
COMMON_SERVER_ARGS,
TestDeterministicBase,
)
register_cuda_ci(est_time=200, suite="nightly-4-gpu", nightly=True)
QWEN3_NEXT = "Qwen/Qwen3-Next-80B-A3B-Instruct"
class TestFlashInferDeterministic(TestDeterministicBase):
@classmethod
def get_model(cls):
return QWEN3_NEXT
# Test with flashinfer attention backend
@classmethod
def get_server_args(cls):
args = COMMON_SERVER_ARGS
args.extend(["--attention-backend", "flashinfer", "--tp", "4"])
return args
class TestTritonDeterministic(TestDeterministicBase):
@classmethod
def get_model(cls):
return QWEN3_NEXT
# Test with triton attention backend
@classmethod
def get_server_args(cls):
args = COMMON_SERVER_ARGS
args.extend(["--attention-backend", "triton", "--tp", "4"])
return args
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,138 @@
import json
import unittest
import warnings
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.run_eval import run_eval
from sglang.test.test_utils import (
DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1,
DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2,
DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1,
DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2,
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
ModelLaunchSettings,
check_evaluation_test_results,
parse_models,
popen_launch_server,
write_results_to_json,
)
register_cuda_ci(est_time=3600, suite="nightly-eval-text-2-gpu", nightly=True)
MODEL_SCORE_THRESHOLDS = {
"meta-llama/Llama-3.1-8B-Instruct": 0.82,
"mistralai/Mistral-7B-Instruct-v0.3": 0.58,
"deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct": 0.85,
"google/gemma-2-27b-it": 0.91,
"meta-llama/Llama-3.1-70B-Instruct": 0.95,
"mistralai/Mixtral-8x7B-Instruct-v0.1": 0.616,
"Qwen/Qwen2-57B-A14B-Instruct": 0.86,
"neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8": 0.83,
"neuralmagic/Mistral-7B-Instruct-v0.3-FP8": 0.54,
"neuralmagic/DeepSeek-Coder-V2-Lite-Instruct-FP8": 0.835,
"zai-org/GLM-4.5-Air-FP8": 0.75,
# The threshold of neuralmagic/gemma-2-2b-it-FP8 should be 0.6, but this model has some accuracy regression.
# The fix is tracked at https://github.com/sgl-project/sglang/issues/4324, we set it to 0.50, for now, to make CI green.
"neuralmagic/gemma-2-2b-it-FP8": 0.50,
"neuralmagic/Meta-Llama-3.1-70B-Instruct-FP8": 0.94,
"neuralmagic/Mixtral-8x7B-Instruct-v0.1-FP8": 0.65,
"neuralmagic/Qwen2-72B-Instruct-FP8": 0.94,
"neuralmagic/Qwen2-57B-A14B-Instruct-FP8": 0.82,
}
# Do not use `CustomTestCase` since `test_mgsm_en_all_models` does not want retry
class TestNightlyGsm8KEval(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.models = []
models_tp1 = parse_models(
DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1
) + parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1)
for model_path in models_tp1:
cls.models.append(ModelLaunchSettings(model_path, tp_size=1))
models_tp2 = parse_models(
DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2
) + parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2)
for model_path in models_tp2:
cls.models.append(ModelLaunchSettings(model_path, tp_size=2))
cls.base_url = DEFAULT_URL_FOR_TEST
def test_mgsm_en_all_models(self):
warnings.filterwarnings(
"ignore", category=ResourceWarning, message="unclosed.*socket"
)
is_first = True
all_results = []
for model_setup in self.models:
with self.subTest(model=model_setup.model_path):
other_args = list(model_setup.extra_args)
error_message = None
if model_setup.model_path == "meta-llama/Llama-3.1-70B-Instruct":
other_args.extend(["--mem-fraction-static", "0.9"])
process = popen_launch_server(
model=model_setup.model_path,
other_args=other_args,
base_url=self.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
)
try:
args = SimpleNamespace(
base_url=self.base_url,
model=model_setup.model_path,
eval_name="mgsm_en",
num_examples=None,
num_threads=1024,
)
metrics = run_eval(args)
print(
f"{'=' * 42}\n{model_setup.model_path} - metrics={metrics} score={metrics['score']}\n{'=' * 42}\n"
)
write_results_to_json(
model_setup.model_path, metrics, "w" if is_first else "a"
)
is_first = False
# 0.0 for empty latency, None for no error
all_results.append(
(model_setup.model_path, metrics["score"], 0.0, error_message)
)
except Exception as e:
# Capture error message for the summary table
error_message = str(e)
# Still append result with error info (use None for N/A metrics to match else clause)
all_results.append(
(model_setup.model_path, None, None, error_message)
)
print(f"Error evaluating {model_setup.model_path}: {error_message}")
finally:
kill_process_tree(process.pid)
try:
with open("results.json", "r") as f:
print("\nFinal Results from results.json:")
print(json.dumps(json.load(f), indent=2))
except Exception as e:
print(f"Error reading results.json: {e}")
# Check all scores after collecting all results
check_evaluation_test_results(
all_results,
self.__class__.__name__,
model_accuracy_thresholds=MODEL_SCORE_THRESHOLDS,
model_count=len(self.models),
)
if __name__ == "__main__":
unittest.main()
+145
View File
@@ -0,0 +1,145 @@
import json
import unittest
import warnings
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.run_eval import run_eval
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
ModelEvalMetrics,
ModelLaunchSettings,
check_evaluation_test_results,
popen_launch_server,
write_results_to_json,
)
register_cuda_ci(est_time=7200, suite="nightly-eval-vlm-2-gpu", nightly=True)
MODEL_THRESHOLDS = {
# Conservative thresholds on 100 MMMU samples, especially for latency thresholds
ModelLaunchSettings("deepseek-ai/deepseek-vl2-small"): ModelEvalMetrics(
0.320, 56.1
),
ModelLaunchSettings("deepseek-ai/Janus-Pro-7B"): ModelEvalMetrics(0.285, 40.3),
ModelLaunchSettings("Efficient-Large-Model/NVILA-8B-hf"): ModelEvalMetrics(
0.270, 56.7
),
ModelLaunchSettings("Efficient-Large-Model/NVILA-Lite-2B-hf"): ModelEvalMetrics(
0.270, 23.8
),
ModelLaunchSettings("google/gemma-3-4b-it"): ModelEvalMetrics(0.360, 10.9),
ModelLaunchSettings("google/gemma-3n-E4B-it"): ModelEvalMetrics(0.270, 17.7),
ModelLaunchSettings("mistral-community/pixtral-12b"): ModelEvalMetrics(0.360, 16.6),
ModelLaunchSettings("moonshotai/Kimi-VL-A3B-Instruct"): ModelEvalMetrics(
0.330, 22.3
),
ModelLaunchSettings("openbmb/MiniCPM-o-2_6"): ModelEvalMetrics(0.330, 29.3),
ModelLaunchSettings("openbmb/MiniCPM-v-2_6"): ModelEvalMetrics(0.259, 36.3),
ModelLaunchSettings("OpenGVLab/InternVL2_5-2B"): ModelEvalMetrics(0.300, 17.0),
ModelLaunchSettings("Qwen/Qwen2-VL-7B-Instruct"): ModelEvalMetrics(0.310, 83.3),
ModelLaunchSettings("Qwen/Qwen2.5-VL-7B-Instruct"): ModelEvalMetrics(0.340, 31.9),
ModelLaunchSettings(
"Qwen/Qwen3-VL-30B-A3B-Instruct", extra_args=["--tp=2"]
): ModelEvalMetrics(0.29, 37.0),
ModelLaunchSettings(
"unsloth/Mistral-Small-3.1-24B-Instruct-2503"
): ModelEvalMetrics(0.310, 16.7),
ModelLaunchSettings("XiaomiMiMo/MiMo-VL-7B-RL"): ModelEvalMetrics(0.28, 32.0),
ModelLaunchSettings("zai-org/GLM-4.1V-9B-Thinking"): ModelEvalMetrics(0.280, 30.4),
ModelLaunchSettings(
"zai-org/GLM-4.5V-FP8", extra_args=["--tp=2"]
): ModelEvalMetrics(0.26, 32.0),
}
class TestNightlyVLMMmmuEval(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.models = list(MODEL_THRESHOLDS.keys())
cls.base_url = DEFAULT_URL_FOR_TEST
def test_mmmu_vlm_models(self):
warnings.filterwarnings(
"ignore", category=ResourceWarning, message="unclosed.*socket"
)
is_first = True
all_results = []
for model in self.models:
model_path = model.model_path
error_message = None
with self.subTest(model=model_path):
process = popen_launch_server(
model=model_path,
base_url=self.base_url,
other_args=model.extra_args,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
)
try:
args = SimpleNamespace(
base_url=self.base_url,
model=model_path,
eval_name="mmmu",
num_examples=100,
num_threads=64,
max_tokens=30,
)
args.return_latency = True
metrics, latency = run_eval(args)
metrics["score"] = round(metrics["score"], 4)
metrics["latency"] = round(latency, 4)
print(
f"{'=' * 42}\n{model_path} - metrics={metrics} score={metrics['score']}\n{'=' * 42}\n"
)
write_results_to_json(model_path, metrics, "w" if is_first else "a")
is_first = False
all_results.append(
(
model_path,
metrics["score"],
metrics["latency"],
error_message,
)
)
except Exception as e:
# Capture error message for the summary table
error_message = str(e)
# Still append result with error info (use None for N/A metrics to match else clause)
all_results.append((model_path, None, None, error_message))
print(f"Error evaluating {model_path}: {error_message}")
finally:
kill_process_tree(process.pid)
try:
with open("results.json", "r") as f:
print("\nFinal Results from results.json:")
print(json.dumps(json.load(f), indent=2))
except Exception as e:
print(f"Error reading results: {e}")
model_accuracy_thresholds = {
model.model_path: threshold.accuracy
for model, threshold in MODEL_THRESHOLDS.items()
}
model_latency_thresholds = {
model.model_path: threshold.eval_time
for model, threshold in MODEL_THRESHOLDS.items()
}
check_evaluation_test_results(
all_results,
self.__class__.__name__,
model_accuracy_thresholds=model_accuracy_thresholds,
model_latency_thresholds=model_latency_thresholds,
)
if __name__ == "__main__":
unittest.main()
+458
View File
@@ -0,0 +1,458 @@
# SPDX-License-Identifier: Apache-2.0
from typing import Callable
import pytest
import torch
from flashinfer import fp4_quantize, scaled_fp4_grouped_quantize
from sglang.test.ci.ci_register import register_cuda_ci
register_cuda_ci(est_time=300, suite="nightly-4-gpu-b200", nightly=True)
from flashinfer.fused_moe import cutlass_fused_moe as flashinfer_cutlass_fused_moe
from sgl_kernel import scaled_fp4_quant, silu_and_mul
from torch.nn import functional as F
from sglang.srt.layers.moe.cutlass_moe import cutlass_moe_fp4
from sglang.srt.layers.moe.cutlass_moe_params import CutlassMoEParams, CutlassMoEType
from sglang.srt.layers.moe.topk import TopKConfig, select_experts
if torch.cuda.get_device_capability() < (10, 0):
pytest.skip(
reason="Nvfp4 Requires compute capability of 10 or above.",
allow_module_level=True,
)
kE2M1ToFloat = torch.tensor(
[0.0, 0.5, 1.0, 1.5, 2.0, 3.0, 4.0, 6.0], dtype=torch.float32
)
FLOAT8_E4M3_MAX = 448.0
FLOAT4_E2M1_MAX = 6.0
def convert_swizzled_to_linear(a_sf_swizzled: torch.Tensor, m, k, block_size):
m_tiles = (m + 128 - 1) // 128
f = block_size * 4
k_tiles = (k + f - 1) // f
tmp = torch.reshape(a_sf_swizzled, (1, m_tiles, k_tiles, 32, 4, 4))
tmp = torch.permute(tmp, (0, 1, 4, 3, 2, 5))
out = tmp.reshape(m_tiles * 128, k_tiles * f // block_size)
return out[0:m, 0:k]
def dequantize_nvfp4_to_dtype(
tensor_fp4, tensor_sf, global_scale, dtype, device, block_size=16
):
"""Dequantize the fp4 tensor back to high precision."""
# Two fp4 values are packed into one uint8.
assert tensor_fp4.dtype == torch.uint8
m, packed_k = tensor_fp4.shape
k = packed_k * 2
tensor_f32 = break_fp4_bytes(tensor_fp4, dtype)
tensor_f32 = tensor_f32.reshape(m, k // block_size, block_size)
tensor_sf = tensor_sf.view(torch.float8_e4m3fn)
tensor_sf = convert_swizzled_to_linear(tensor_sf, m, k, block_size)
tensor_sf_dtype = tensor_sf.to(torch.float32) / global_scale
# scale the tensor
out = (tensor_f32 * tensor_sf_dtype.unsqueeze(-1)).reshape(m, k)
return out.to(dtype=dtype)
def break_fp4_bytes(a, dtype):
assert a.dtype == torch.uint8
m, n = a.shape
# Vectorized nibble processing
a_flat = a.flatten()
high = (a_flat & 0xF0) >> 4 # Upper nibbles
low = a_flat & 0x0F # Lower nibbles
# Combine nibbles for batch processing
combined = torch.stack((low, high), dim=1).flatten()
# Vectorized sign and magnitude extraction
signs = (combined & 0x08).to(torch.bool) # Sign bits
abs_vals = (combined & 0x07).to(torch.long) # Magnitude indices
# Device-aware lookup and sign application
kE2M1 = kE2M1ToFloat.to(device=a.device)
values = kE2M1[abs_vals] * torch.where(signs, -1.0, 1.0)
# Reshape to final form
return values.reshape(m, n * 2).to(dtype=dtype)
def compute_routing(router_logits: torch.Tensor, top_k: int):
routing_weights = torch.softmax(router_logits, dim=1, dtype=torch.float)
routing_weights, selected_experts = torch.topk(routing_weights, top_k, dim=-1)
routing_weights /= routing_weights.sum(dim=-1, keepdim=True)
routing_weights = routing_weights.float()
return routing_weights, selected_experts
def prepare_inputs(
hidden_states: torch.Tensor,
router_logits: torch.Tensor,
num_experts: int,
topk: int,
):
routing_weights, topk_idx = compute_routing(router_logits, topk)
masked_m = []
for i in range(num_experts):
mask = topk_idx.view(-1) == i
masked_m.append(mask.sum())
masked_m = torch.tensor(masked_m, dtype=torch.int32)
hidden_states_3d = torch.empty(
(num_experts, max(masked_m), hidden_states.shape[1]), dtype=hidden_states.dtype
)
for i in range(num_experts):
hidden_states_3d[i, : masked_m[i], :] = hidden_states[topk_idx.view(-1) == i]
return hidden_states_3d, masked_m, topk_idx, routing_weights
MNK_FACTORS = [
(2, 1024, 1024),
(2, 1024, 1536),
(2, 3072, 1024),
(2, 3072, 1536),
(64, 1024, 1024),
(64, 1024, 1536),
(64, 3072, 1024),
(64, 2048, 1024),
(224, 1024, 1024),
(224, 1024, 1536),
]
# Reference implementation of torch_moe
def torch_moe(a, w1, w2, score, topk, expert_map):
B, D = a.shape
a = a.view(B, -1, D).repeat(1, topk, 1).reshape(-1, D)
out = torch.zeros(B * topk, w2.shape[1], dtype=a.dtype, device=a.device)
score = torch.softmax(score, dim=-1, dtype=torch.float32)
topk_weight, topk_ids = torch.topk(score, topk)
topk_weight = topk_weight.view(-1)
topk_ids = topk_ids.view(-1)
if expert_map is not None:
topk_ids = expert_map[topk_ids]
for i in range(w1.shape[0]):
mask = topk_ids == i
if mask.sum():
out[mask] = silu_and_mul(a[mask] @ w1[i].transpose(0, 1)) @ w2[i].transpose(
0, 1
)
return (
out.view(B, -1, w2.shape[1]) * topk_weight.view(B, -1, 1).to(out.dtype)
).sum(dim=1)
def torch_moe_nvfp4(a, w1, w2, topk, topk_weight, topk_ids):
B, D = a.shape
a = a.view(B, -1, D).repeat(1, topk, 1).reshape(-1, D)
out = torch.zeros(B * topk, w2.shape[1], dtype=a.dtype, device=a.device)
topk_weight = topk_weight.view(-1)
topk_ids = topk_ids.view(-1)
for i in range(w1.shape[0]):
mask = topk_ids == i
if mask.sum():
m = w1[i].shape[0]
assert m % 2 == 0
# Note: w1 and w3 are swapped!
w3_expert, w1_expert = w1[i][m // 2 :, :], w1[i][: m // 2, :]
inter = F.silu(a[mask] @ w1_expert.t()) * (a[mask] @ w3_expert.t())
inter_gs = torch.tensor(1.0).cuda()
inter_q, inter_blockscale = fp4_quantize(inter, inter_gs)
inter = dequantize_nvfp4_to_dtype(
inter_q,
inter_blockscale,
inter_gs,
dtype=inter.dtype,
device=inter.device,
block_size=16,
).cuda()
out[mask] = inter @ w2[i].transpose(0, 1)
return (
out.view(B, -1, w2.shape[1]) * topk_weight.view(B, -1, 1).to(out.dtype)
).sum(dim=1)
def flashinfer_cutedsl_grouped_gemm_nt_masked(
hidden_states: torch.Tensor, # 3d
input_global_scale: torch.Tensor, # (l,)
weights: torch.Tensor,
w_global_scale: torch.Tensor, # (l,)
masked_m: torch.Tensor,
):
from flashinfer.cute_dsl.blockscaled_gemm import grouped_gemm_nt_masked
# hidden_states: [l, m, k]
# weights: [l, n, k]
aq, aq_sf = scaled_fp4_grouped_quantize(
hidden_states,
masked_m.to(hidden_states.device),
input_global_scale,
)
num_experts, n, k = weights.shape
bq, bq_sf = scaled_fp4_grouped_quantize(
weights,
torch.ones(num_experts, device=weights.device, dtype=torch.int32) * n,
w_global_scale,
)
out = torch.zeros(
(num_experts, max(masked_m), n), dtype=weights.dtype, device=aq.device
)
out = out.permute(1, 2, 0) # requirement of kernel
sf_vec_size = 16
ab_dtype = "float4_e2m1fn"
sf_dtype = "float8_e4m3fn"
c_dtype = "bfloat16"
alpha = 1.0 / (input_global_scale * w_global_scale).to(out.dtype).view(
1, 1, num_experts
)
def get_cute_dtype(input: torch.Tensor) -> str:
if input.dtype == torch.bfloat16:
return "bfloat16"
elif input.dtype == torch.float16:
return "float16"
elif input.dtype == torch.float32:
return "float32"
else:
raise ValueError(f"Unsupported cute dtype {input.dtype}")
grouped_gemm_nt_masked(
(aq, aq_sf),
(bq, bq_sf),
out,
masked_m.to(aq.device),
ab_dtype=ab_dtype,
sf_dtype=sf_dtype,
c_dtype=c_dtype,
sf_vec_size=sf_vec_size,
alpha=alpha,
alpha_dtype=get_cute_dtype(alpha),
)
return out
def check_moe(
m: int,
n: int,
k: int,
e: int,
topk: int,
dtype: torch.dtype,
moe_impl: Callable,
flip_w13: bool,
):
torch.manual_seed(7)
a = torch.randn((m, k), device="cuda", dtype=dtype) / 10
w1 = torch.randn((e, 2 * n, k), device="cuda", dtype=dtype) / 10
quant_blocksize = 16
round_up = lambda x, y: (x + y - 1) // y * y
sf_w1_2n = round_up(2 * n, 128)
sf_w1_k = round_up(k // quant_blocksize, 4)
w1_blockscale = torch.empty(
(e, sf_w1_2n, sf_w1_k), device="cuda", dtype=torch.float8_e4m3fn
)
w2 = torch.randn((e, k, n), device="cuda", dtype=dtype) / 10
sf_w2_k = round_up(k, 128)
sf_w2_n = round_up(n // quant_blocksize, 4)
w2_blockscale = torch.empty(
(e, sf_w2_k, sf_w2_n), device="cuda", dtype=torch.float8_e4m3fn
)
w1_q = torch.empty((e, 2 * n, k // 2), device="cuda", dtype=torch.uint8)
w2_q = torch.empty((e, k, n // 2), device="cuda", dtype=torch.uint8)
w1_gs = torch.empty((e,), device="cuda", dtype=torch.float32)
w2_gs = torch.empty((e,), device="cuda", dtype=torch.float32)
for expert in range(e):
w1_amax = torch.abs(w1).max().to(torch.float32)
w2_amax = torch.abs(w2).max().to(torch.float32)
w1_gs[expert] = FLOAT8_E4M3_MAX * FLOAT4_E2M1_MAX / w1_amax
w2_gs[expert] = FLOAT8_E4M3_MAX * FLOAT4_E2M1_MAX / w2_amax
w1_q[expert], w1_blockscale[expert] = scaled_fp4_quant(
w1[expert], w1_gs[expert]
)
w2_q[expert], w2_blockscale[expert] = scaled_fp4_quant(
w2[expert], w2_gs[expert]
)
score = torch.randn((m, e), device="cuda", dtype=dtype)
topk_output = select_experts(
hidden_states=a,
router_logits=score,
topk_config=TopKConfig(top_k=topk, renormalize=False),
)
topk_weights, topk_ids, _ = topk_output
a1_gs = torch.ones((e,), device="cuda", dtype=torch.float32)
a2_gs = torch.ones((e,), device="cuda", dtype=torch.float32)
test_output = moe_impl(
a=a,
topk_weights=topk_weights,
topk_ids=topk_ids,
w1_q=w1_q,
w2_q=w2_q,
a1_gs=a1_gs,
w1_blockscale=w1_blockscale,
w1_alphas=(1 / w1_gs),
a2_gs=a2_gs,
w2_blockscale=w2_blockscale,
w2_alphas=(1 / w2_gs),
)
# Reference check:
a_global_scale = (
(FLOAT8_E4M3_MAX * FLOAT4_E2M1_MAX) / torch.amax(a.flatten(), dim=-1)
).to(torch.float32)
a_fp4, a_scale_interleaved = scaled_fp4_quant(a, a_global_scale)
_, m_k = a_fp4.shape
a_in_dtype = dequantize_nvfp4_to_dtype(
a_fp4,
a_scale_interleaved,
a_global_scale,
dtype=a.dtype,
device=a.device,
block_size=quant_blocksize,
)
w1_d = torch.empty((e, 2 * n, k), device="cuda", dtype=dtype)
w2_d = torch.empty((e, k, n), device="cuda", dtype=dtype)
for idx in range(0, e):
w1_d[idx] = dequantize_nvfp4_to_dtype(
w1_q[idx],
w1_blockscale[idx],
w1_gs[idx],
dtype=w1.dtype,
device=w1.device,
block_size=quant_blocksize,
)
w2_d[idx] = dequantize_nvfp4_to_dtype(
w2_q[idx],
w2_blockscale[idx],
w2_gs[idx],
dtype=w2.dtype,
device=w2.device,
block_size=quant_blocksize,
)
if flip_w13:
dim = -2
size = w1_d.size(dim)
assert size % 2 == 0, f"Expected even size in dim {dim}, got {size}"
half = size // 2
# Reorder weight
w1, w3 = w1_d.split(half, dim=dim)
w1_d = torch.cat([w3, w1], dim=dim).contiguous()
torch_output = torch_moe(a_in_dtype, w1_d, w2_d, score, topk, None)
torch.testing.assert_close(torch_output, test_output, atol=1e-1, rtol=1e-1)
@pytest.mark.parametrize("m,n,k", MNK_FACTORS)
@pytest.mark.parametrize("e", [40, 64, 256])
@pytest.mark.parametrize("topk", [1, 6, 8])
@pytest.mark.parametrize("dtype", [torch.half, torch.bfloat16])
@torch.inference_mode()
def test_cutlass_fp4_moe_no_graph(
m: int, n: int, k: int, e: int, topk: int, dtype: torch.dtype
):
def cutlass_moe_impl(
a,
topk_weights,
topk_ids,
w1_q,
w2_q,
a1_gs,
w1_blockscale,
w1_alphas,
a2_gs,
w2_blockscale,
w2_alphas,
):
params = CutlassMoEParams(
CutlassMoEType.BlockscaledFP4,
device=a.device,
num_experts=e,
intermediate_size_per_partition=n, # n
hidden_size=k,
) # k
return cutlass_moe_fp4(
a=a,
a1_gscale=a1_gs,
w1_fp4=w1_q,
w1_blockscale=w1_blockscale,
w1_alphas=w1_alphas,
a2_gscale=a2_gs,
w2_fp4=w2_q,
w2_blockscale=w2_blockscale,
w2_alphas=w2_alphas,
topk_weights=topk_weights,
topk_ids=topk_ids,
params=params,
apply_router_weight_on_input=False,
)
check_moe(m, n, k, e, topk, dtype, cutlass_moe_impl, flip_w13=False)
@pytest.mark.parametrize("m,n,k", MNK_FACTORS)
@pytest.mark.parametrize("e", [40, 64, 256])
@pytest.mark.parametrize("topk", [1, 6, 8])
@pytest.mark.parametrize("dtype", [torch.half, torch.bfloat16])
@torch.inference_mode()
def test_flashinfer_fp4_moe_no_graph(
m: int, n: int, k: int, e: int, topk: int, dtype: torch.dtype
):
def flashinfer_moe_impl(
a,
topk_weights,
topk_ids,
w1_q,
w2_q,
a1_gs,
w1_blockscale,
w1_alphas,
a2_gs,
w2_blockscale,
w2_alphas,
):
return flashinfer_cutlass_fused_moe(
a,
topk_ids.to(torch.int),
topk_weights,
w1_q.view(torch.long),
w2_q.view(torch.long),
a.dtype,
quant_scales=[
a1_gs,
w1_blockscale.view(torch.int32),
w1_alphas,
a2_gs,
w2_blockscale.view(torch.int32),
w2_alphas,
],
)[0]
check_moe(m, n, k, e, topk, dtype, flashinfer_moe_impl, flip_w13=True)
if __name__ == "__main__":
test_cutlass_fp4_moe_no_graph(224, 1024, 1024, 256, 8, torch.half)
test_flashinfer_fp4_moe_no_graph(224, 1024, 1024, 256, 8, torch.half)
+619
View File
@@ -0,0 +1,619 @@
import unittest
from typing import Optional
from unittest.mock import MagicMock, patch
import torch
from sglang.test.ci.ci_register import register_cuda_ci
register_cuda_ci(est_time=2, suite="nightly-1-gpu", nightly=True)
from sglang.srt.layers import dp_attention as _dp_attn
# Patch DP-attention globals before importing backends
_dp_attn.get_attention_tp_size = lambda: 1 # TP size = 1 for unit test
from sglang.srt.configs.model_config import AttentionArch
from sglang.srt.layers.attention.nsa.nsa_indexer import (
BaseIndexerMetadata,
Indexer,
rotate_activation,
)
from sglang.srt.layers.attention.nsa_backend import NativeSparseAttnBackend
from sglang.srt.layers.layernorm import LayerNorm
from sglang.srt.layers.linear import LinearBase
from sglang.srt.mem_cache.memory_pool import NSATokenToKVPool
from sglang.srt.model_executor.forward_batch_info import ForwardBatch, ForwardMode
from sglang.srt.server_args import ServerArgs, set_global_server_args_for_scheduler
from sglang.test.test_utils import CustomTestCase
# Global configuration for all indexer tests
DEFAULT_CONFIG = {
"device": "cuda",
"dtype": torch.bfloat16,
"kv_cache_dtype": torch.float8_e4m3fn,
"context_len": 2048,
"max_bs": 64,
"hidden_size": 5120,
"index_n_heads": 1,
"index_head_dim": 128,
"rope_head_dim": 64,
"index_topk": 64,
"q_lora_rank": 1536,
"kv_lora_rank": 512,
"qk_rope_head_dim": 64,
"max_position_embeddings": 163840,
"rope_theta": 10000.0,
"layer_id": 0,
"page_size": 64,
}
class MockIndexerMetadata(BaseIndexerMetadata):
"""Mock implementation of BaseIndexerMetadata for testing."""
def __init__(self, batch_size, seq_lens, page_table=None):
self.batch_size = batch_size
self.seq_lens = seq_lens
self.page_table = page_table
self.device = "cuda"
def get_seqlens_int32(self) -> torch.Tensor:
"""Return: (batch_size,) int32 tensor"""
return torch.tensor(self.seq_lens, dtype=torch.int32, device=self.device)
def get_page_table_64(self) -> torch.Tensor:
"""Return: (batch_size, num_blocks) int32, page table with page size 64."""
if self.page_table is not None:
return self.page_table
# Create a simple page table for testing
max_seq_len = max(self.seq_lens)
num_blocks = (max_seq_len + 63) // 64 # Round up to page size 64
page_table = torch.zeros(
(self.batch_size, num_blocks), dtype=torch.int32, device=self.device
)
for i in range(self.batch_size):
# Simple linear mapping: block i maps to page i
num_blocks_needed = (self.seq_lens[i] + 63) // 64
page_table[i, :num_blocks_needed] = torch.arange(
num_blocks_needed, device=self.device
)
return page_table
def get_seqlens_expanded(self) -> torch.Tensor:
"""Return: (sum_extend_seq_len,) int32 tensor"""
# For extend mode, each new token attends to progressively more tokens
# For a sequence being extended from position 0 to seq_len, token i attends to i+1 tokens
result = []
for seq_len in self.seq_lens:
result.extend(range(1, seq_len + 1))
return torch.tensor(result, dtype=torch.int32, device=self.device)
def topk_transform(
self,
logits: torch.Tensor,
topk: int,
ks: Optional[torch.Tensor] = None,
) -> torch.Tensor:
"""
Perform topk selection on the logits.
For testing, just return the topk indices.
"""
return torch.topk(logits, k=topk, dim=-1).indices
class MockModelRunner:
def __init__(self, config=None):
self.device = "cuda"
self.config = {**DEFAULT_CONFIG, **(config or {})}
self.dtype = self.config["dtype"]
self.kv_cache_dtype = self.config["kv_cache_dtype"]
self.is_hybrid_swa = False
# Model configuration
attention_arch = AttentionArch.MLA
max_context_len = self.config["context_len"]
max_batch_size = self.config["max_bs"]
# Create mock hf_config for NSA - instantiate it as an object, not a type
hf_config = type(
"HfConfig",
(),
{
"architectures": ["DeepseekV3ForCausalLM"],
"index_topk": self.config["index_topk"],
"index_head_dim": self.config["index_head_dim"],
"index_n_heads": self.config["index_n_heads"],
},
)()
self.model_config = type(
"ModelConfig",
(),
{
"context_len": max_context_len,
"is_multimodal": False,
"attention_arch": attention_arch,
"num_attention_heads": 128,
"kv_lora_rank": self.config["kv_lora_rank"],
"qk_rope_head_dim": self.config["qk_rope_head_dim"],
"hf_config": hf_config,
},
)()
self.sliding_window_size = None
self.page_size = self.config["page_size"]
# Create req_to_token_pool
self.req_to_token_pool = type(
"TokenPool",
(),
{
"size": max_batch_size,
"req_to_token": torch.zeros(
max_batch_size,
max_context_len,
dtype=torch.int32,
device=self.device,
),
},
)()
# Create NSATokenToKVPool
max_total_num_tokens = max_batch_size * max_context_len
self.token_to_kv_pool = NSATokenToKVPool(
size=max_total_num_tokens,
page_size=self.config["page_size"],
dtype=self.config["kv_cache_dtype"],
kv_lora_rank=self.config["kv_lora_rank"],
qk_rope_head_dim=self.config["qk_rope_head_dim"],
layer_num=1,
device=self.device,
index_head_dim=self.config["index_head_dim"],
enable_memory_saver=False,
)
# Required by backend with NSA-specific attributes
self.server_args = type(
"ServerArgs",
(),
{
"kv_cache_dtype": "auto",
"speculative_eagle_topk": None,
"speculative_num_draft_tokens": 0,
"enable_deterministic_inference": False,
"nsa_prefill_backend": "flashmla_sparse",
"nsa_decode_backend": "fa3",
},
)()
@unittest.skipIf(not torch.cuda.is_available(), "Test requires CUDA")
class TestNSAIndexer(CustomTestCase):
@classmethod
def setUpClass(cls):
"""Set up global server args for testing."""
server_args = ServerArgs(model_path="dummy")
server_args.enable_dp_attention = False
server_args.nsa_prefill_backend = "flashmla_sparse"
server_args.nsa_decode_backend = "flashmla_sparse"
set_global_server_args_for_scheduler(server_args)
# Check GPU capability for FP8
if torch.cuda.is_available():
compute_capability = torch.cuda.get_device_capability()
cls.supports_fp8 = compute_capability[0] >= 9 # Hopper or newer
@classmethod
def tearDownClass(cls):
"""Clean up after all tests."""
pass
def setUp(self):
# Test parameters
self.batch_size = 2
self.seq_len = 128
self.config = DEFAULT_CONFIG.copy()
self.device = "cuda"
self.dtype = torch.bfloat16
def _init_model_runner(self, config_override=None):
"""Initialize model runner with optional config override."""
config = self.config.copy()
if config_override:
config.update(config_override)
self.model_runner = MockModelRunner(config)
self.backend = NativeSparseAttnBackend(self.model_runner)
def _create_indexer(self, **kwargs):
"""Create an Indexer instance with default parameters."""
params = {
"hidden_size": self.config["hidden_size"],
"index_n_heads": self.config["index_n_heads"],
"index_head_dim": self.config["index_head_dim"],
"rope_head_dim": self.config["rope_head_dim"],
"index_topk": self.config["index_topk"],
"q_lora_rank": self.config["q_lora_rank"],
"max_position_embeddings": self.config["max_position_embeddings"],
"rope_theta": self.config["rope_theta"],
"layer_id": self.config["layer_id"],
"scale_fmt": "ue8m0",
"block_size": 128,
"quant_config": None, # No quantization for testing
}
params.update(kwargs)
torch.set_default_dtype(self.dtype)
indexer = Indexer(**params)
# Move indexer to CUDA device
indexer = indexer.to(device=self.device)
# Convert linear layer weights to bfloat16 (but preserve LayerNorm's float32
# and weights_proj's float32 - it uses params_dtype=torch.float32 in production)
# Need to recursively convert LinearBase submodules (like ReplicatedLinear)
for name, module in indexer.named_modules():
# Check for LinearBase (parent of ReplicatedLinear) but exclude LayerNorm
# Also exclude weights_proj which uses float32 params in production
if isinstance(module, LinearBase) and not isinstance(module, LayerNorm):
if "weights_proj" not in name:
module.to(dtype=self.dtype)
return indexer
def _create_forward_batch(
self, mode, batch_size=None, seq_len=None, extend_len=None
):
"""Create a forward batch for testing."""
batch_size = batch_size or self.batch_size
seq_len = seq_len or self.seq_len
if mode == ForwardMode.EXTEND:
q_len = extend_len or seq_len
total_len = seq_len
forward_batch = ForwardBatch(
batch_size=batch_size,
input_ids=torch.randint(
0, 100, (batch_size, q_len), device=self.device
),
out_cache_loc=torch.arange(
batch_size * (total_len - q_len),
batch_size * total_len,
device=self.device,
),
seq_lens_sum=batch_size * total_len,
forward_mode=mode,
req_pool_indices=torch.arange(batch_size, device=self.device),
seq_lens=torch.tensor([total_len] * batch_size, device=self.device),
seq_lens_cpu=torch.tensor([total_len] * batch_size, device="cpu"),
extend_prefix_lens=torch.tensor(
[total_len - q_len] * batch_size, device=self.device
),
extend_prefix_lens_cpu=torch.tensor(
[total_len - q_len] * batch_size, device="cpu"
),
extend_seq_lens=torch.tensor([q_len] * batch_size, device=self.device),
extend_seq_lens_cpu=torch.tensor([q_len] * batch_size, device="cpu"),
attn_backend=self.backend,
)
else: # ForwardMode.DECODE
decode_len = 1
total_len = seq_len + decode_len
forward_batch = ForwardBatch(
batch_size=batch_size,
input_ids=torch.randint(
0, 100, (batch_size, decode_len), device=self.device
),
out_cache_loc=torch.arange(
batch_size * seq_len, batch_size * total_len, device=self.device
),
seq_lens_sum=batch_size * total_len,
forward_mode=mode,
req_pool_indices=torch.arange(batch_size, device=self.device),
seq_lens=torch.tensor([total_len] * batch_size, device=self.device),
seq_lens_cpu=torch.tensor([total_len] * batch_size, device="cpu"),
attn_backend=self.backend,
)
# Add token pools
forward_batch.req_to_token_pool = self.model_runner.req_to_token_pool
forward_batch.token_to_kv_pool = self.model_runner.token_to_kv_pool
# Mock write to req_to_token_pool
page_size = self.model_runner.page_size
for i in range(batch_size):
seq_length = total_len
for j in range(seq_length):
self.model_runner.req_to_token_pool.req_to_token[i, j] = (
i * seq_length + j + page_size
)
return forward_batch
def _verify_topk_output(self, topk_indices, batch_size, q_len, topk):
"""Verify the topk indices output shape and basic properties."""
self.assertIsNotNone(topk_indices)
self.assertEqual(topk_indices.device.type, "cuda")
# Check shape - should be (total_q_len, topk_padded)
# where topk_padded is aligned to 2048
self.assertEqual(len(topk_indices.shape), 2)
self.assertEqual(topk_indices.shape[0], batch_size * q_len)
# Check that topk is padded to at least topk
self.assertGreaterEqual(topk_indices.shape[1], topk)
# Check for padding values (-1)
has_padding = (topk_indices == -1).any()
self.assertTrue(
has_padding or topk_indices.shape[1] == topk,
"Output should have padding or exact topk size",
)
@patch("sglang.srt.layers.attention.nsa.nsa_indexer.deep_gemm")
def test_indexer_basic_creation(self, mock_deep_gemm):
"""Test basic indexer creation and initialization."""
mock_deep_gemm.get_num_sms.return_value = 132
indexer = self._create_indexer()
self.assertEqual(indexer.hidden_size, self.config["hidden_size"])
self.assertEqual(indexer.n_heads, self.config["index_n_heads"])
self.assertEqual(indexer.head_dim, self.config["index_head_dim"])
self.assertEqual(indexer.rope_head_dim, self.config["rope_head_dim"])
self.assertEqual(indexer.index_topk, self.config["index_topk"])
self.assertEqual(indexer.layer_id, self.config["layer_id"])
@patch("sglang.srt.layers.attention.nsa.nsa_indexer.deep_gemm")
@patch("sglang.srt.layers.attention.nsa.triton_kernel.act_quant")
def test_forward_extend_mode(self, mock_act_quant, mock_deep_gemm):
"""Test indexer forward pass in extend mode."""
if not self.supports_fp8:
self.skipTest("FP8 requires Hopper GPU or newer")
# Setup mocks
mock_deep_gemm.get_num_sms.return_value = 132
mock_deep_gemm.get_paged_mqa_logits_metadata.return_value = MagicMock()
def mock_quant(x, *args, **kwargs):
# Return FP8 tensor and scale
return x.to(torch.float8_e4m3fn), torch.ones(
x.shape[0], dtype=torch.float32, device=x.device
)
mock_act_quant.side_effect = mock_quant
# Mock deep_gemm.fp8_mqa_logits to return logits (ragged path)
def mock_mqa_logits(q, kv, weights, ks, ke, *args, **kwargs):
# q shape: (sum_extend_seq_len, ...), return logits for each query token
num_queries = q.shape[0]
# kv is a tuple (k_fp8, k_scale), get total number of keys from k_fp8
k_fp8, k_scale = kv
max_kv_len = k_fp8.shape[0] # Total keys across all batches (k_offset)
return torch.randn(
num_queries, max_kv_len, dtype=torch.float32, device="cuda"
)
mock_deep_gemm.fp8_mqa_logits.side_effect = mock_mqa_logits
# Also mock the paged version for completeness
def mock_paged_mqa_logits(q, kv, weights, *args, **kwargs):
batch_size = q.shape[0]
seq_len = 128
return torch.randn(batch_size, seq_len, dtype=torch.float32, device="cuda")
mock_deep_gemm.fp8_paged_mqa_logits.side_effect = mock_paged_mqa_logits
self._init_model_runner()
indexer = self._create_indexer()
forward_batch = self._create_forward_batch(ForwardMode.EXTEND)
# Create input tensors
total_tokens = self.batch_size * self.seq_len
hidden_states = torch.randn(
total_tokens,
self.config["hidden_size"],
dtype=self.dtype,
device=self.device,
)
q_lora = torch.randn(
total_tokens,
self.config["q_lora_rank"],
dtype=self.dtype,
device=self.device,
)
positions = torch.arange(total_tokens, device=self.device)
# Run forward pass
with patch.object(
self.backend,
"get_indexer_metadata",
return_value=MockIndexerMetadata(
self.batch_size, [self.seq_len] * self.batch_size
),
):
topk_indices = indexer(
x=hidden_states,
q_lora=q_lora,
positions=positions,
forward_batch=forward_batch,
layer_id=self.config["layer_id"],
)
# Verify output
self._verify_topk_output(
topk_indices, self.batch_size, self.seq_len, self.config["index_topk"]
)
@patch("sglang.srt.layers.attention.nsa.nsa_indexer.deep_gemm")
@patch("sglang.srt.layers.attention.nsa.triton_kernel.act_quant")
def test_forward_decode_mode(self, mock_act_quant, mock_deep_gemm):
"""Test indexer forward pass in decode mode."""
if not self.supports_fp8:
self.skipTest("FP8 requires Hopper GPU or newer")
# Setup mocks
mock_deep_gemm.get_num_sms.return_value = 132
mock_deep_gemm.get_paged_mqa_logits_metadata.return_value = MagicMock()
def mock_quant(x, *args, **kwargs):
return x.to(torch.float8_e4m3fn), torch.ones(
x.shape[0], dtype=torch.float32, device=x.device
)
mock_act_quant.side_effect = mock_quant
def mock_paged_mqa_logits(q, kv, weights, *args, **kwargs):
batch_size = q.shape[0]
seq_len = 128
return torch.randn(batch_size, seq_len, dtype=torch.float32, device="cuda")
mock_deep_gemm.fp8_paged_mqa_logits.side_effect = mock_paged_mqa_logits
self._init_model_runner()
indexer = self._create_indexer()
forward_batch = self._create_forward_batch(ForwardMode.DECODE)
# Create input tensors for decode (batch_size tokens only)
hidden_states = torch.randn(
self.batch_size,
self.config["hidden_size"],
dtype=self.dtype,
device=self.device,
)
q_lora = torch.randn(
self.batch_size,
self.config["q_lora_rank"],
dtype=self.dtype,
device=self.device,
)
positions = torch.arange(self.batch_size, device=self.device)
# Run forward pass
with patch.object(
self.backend,
"get_indexer_metadata",
return_value=MockIndexerMetadata(
self.batch_size, [self.seq_len + 1] * self.batch_size
),
):
topk_indices = indexer(
x=hidden_states,
q_lora=q_lora,
positions=positions,
forward_batch=forward_batch,
layer_id=self.config["layer_id"],
)
# Verify output - decode mode has q_len=1
self._verify_topk_output(
topk_indices, self.batch_size, 1, self.config["index_topk"]
)
def test_rotate_activation(self):
"""Test the Hadamard transform (rotate_activation) function."""
# Test with power-of-2 hidden size
hidden_size = 128
x = torch.randn(16, hidden_size, dtype=torch.bfloat16, device=self.device)
try:
output = rotate_activation(x)
self.assertEqual(output.shape, x.shape)
self.assertEqual(output.dtype, torch.bfloat16)
except ImportError:
self.skipTest("sgl_kernel not available for hadamard_transform")
def test_rotate_activation_invalid_size(self):
"""Test that rotate_activation fails with non-power-of-2 size."""
# Test with non-power-of-2 hidden size
hidden_size = 129 # Not a power of 2
x = torch.randn(16, hidden_size, dtype=torch.bfloat16, device=self.device)
with self.assertRaises(AssertionError):
rotate_activation(x)
def test_indexer_metadata_interface(self):
"""Test the BaseIndexerMetadata interface implementation."""
batch_size = 4
seq_lens = [64, 128, 96, 112]
metadata = MockIndexerMetadata(batch_size, seq_lens)
# Test get_seqlens_int32
seqlens = metadata.get_seqlens_int32()
self.assertEqual(seqlens.shape, (batch_size,))
self.assertEqual(seqlens.dtype, torch.int32)
self.assertTrue(torch.all(seqlens == torch.tensor(seq_lens, device="cuda")))
# Test get_page_table_64
page_table = metadata.get_page_table_64()
self.assertEqual(len(page_table.shape), 2)
self.assertEqual(page_table.shape[0], batch_size)
self.assertEqual(page_table.dtype, torch.int32)
# Test topk_transform
logits = torch.randn(batch_size, 128, device="cuda")
topk = 64
topk_indices = metadata.topk_transform(logits, topk)
self.assertEqual(topk_indices.shape, (batch_size, topk))
# TODO: enable this test after indexer accuracy aligned
# @patch("sglang.srt.layers.attention.nsa.nsa_indexer.deep_gemm")
# def test_indexer_with_different_topk(self, mock_deep_gemm):
# """Test indexer with different topk values."""
# mock_deep_gemm.get_num_sms.return_value = 132
# for topk in [32, 64, 128]:
# with self.subTest(topk=topk):
# indexer = self._create_indexer(index_topk=topk)
# self.assertEqual(indexer.index_topk, topk)
@patch("sglang.srt.layers.attention.nsa.nsa_indexer.deep_gemm")
def test_indexer_with_fused_wk(self, mock_deep_gemm):
"""Test indexer creation with fused wk and weights projection."""
mock_deep_gemm.get_num_sms.return_value = 132
# Note: fuse_wk_and_weights_proj feature is not currently implemented
# This test verifies basic indexer creation still works
indexer = self._create_indexer()
self.assertIsNotNone(indexer)
@patch("sglang.srt.layers.attention.nsa.nsa_indexer.deep_gemm")
def test_indexer_with_alt_stream(self, mock_deep_gemm):
"""Test indexer creation with alternative CUDA stream."""
mock_deep_gemm.get_num_sms.return_value = 132
alt_stream = torch.cuda.Stream()
indexer = self._create_indexer(alt_stream=alt_stream)
self.assertEqual(indexer.alt_stream, alt_stream)
def test_shape_sanity_checks(self):
"""Test various shape combinations for consistency."""
test_configs = [
{"batch_size": 1, "seq_len": 64},
{"batch_size": 4, "seq_len": 128},
{"batch_size": 8, "seq_len": 256},
]
for config in test_configs:
with self.subTest(**config):
batch_size = config["batch_size"]
seq_len = config["seq_len"]
# Test metadata shapes
metadata = MockIndexerMetadata(batch_size, [seq_len] * batch_size)
seqlens = metadata.get_seqlens_int32()
self.assertEqual(seqlens.shape, (batch_size,))
page_table = metadata.get_page_table_64()
expected_blocks = (seq_len + 63) // 64
self.assertEqual(page_table.shape[0], batch_size)
self.assertGreaterEqual(page_table.shape[1], expected_blocks)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,60 @@
import unittest
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.nightly_utils import NightlyBenchmarkRunner
from sglang.test.test_utils import DEFAULT_URL_FOR_TEST
register_cuda_ci(est_time=600, suite="nightly-4-gpu-b200", nightly=True)
PROFILE_DIR = "performance_profiles_gpt_oss_4gpu"
class TestNightlyGptOss4GpuPerformance(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.models = [
(
"openai/gpt-oss-120b",
[
"--tp",
"4",
"--cuda-graph-max-bs",
"200",
"--mem-fraction-static",
"0.93",
],
),
]
cls.base_url = DEFAULT_URL_FOR_TEST
cls.batch_sizes = [1, 1, 8, 16, 64]
cls.input_lens = (4096,)
cls.output_lens = (512,)
cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url)
cls.runner.setup_profile_directory()
def test_bench_one_batch(self):
all_model_succeed = True
for model_path, other_args in self.models:
with self.subTest(model=model_path):
results, success, _ = self.runner.run_benchmark_for_model(
model_path=model_path,
batch_sizes=self.batch_sizes,
input_lens=self.input_lens,
output_lens=self.output_lens,
other_args=other_args,
)
if not success:
all_model_succeed = False
self.runner.add_report(results)
self.runner.write_final_report()
if not all_model_succeed:
raise AssertionError("Some models failed the perf tests.")
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,62 @@
import unittest
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.nightly_utils import NightlyBenchmarkRunner
from sglang.test.test_utils import (
DEFAULT_URL_FOR_TEST,
ModelLaunchSettings,
_parse_int_list_env,
parse_models,
)
register_cuda_ci(est_time=3600, suite="nightly-perf-text-2-gpu", nightly=True)
PROFILE_DIR = "performance_profiles_text_models"
class TestNightlyTextModelsPerformance(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.models = []
# TODO: replace with DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1 or other model lists
for model_path in parse_models("meta-llama/Llama-3.1-8B-Instruct"):
cls.models.append(ModelLaunchSettings(model_path, tp_size=1))
for model_path in parse_models("Qwen/Qwen2-57B-A14B-Instruct"):
cls.models.append(ModelLaunchSettings(model_path, tp_size=2))
# (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1), False, False),
# (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2), False, True),
# (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1), True, False),
# (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2), True, True),
cls.base_url = DEFAULT_URL_FOR_TEST
cls.batch_sizes = [1, 1, 8, 16, 64]
cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096"))
cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512"))
cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url)
cls.runner.setup_profile_directory()
def test_bench_one_batch(self):
all_model_succeed = True
for model_setup in self.models:
with self.subTest(model=model_setup.model_path):
results, success, _ = self.runner.run_benchmark_for_model(
model_path=model_setup.model_path,
batch_sizes=self.batch_sizes,
input_lens=self.input_lens,
output_lens=self.output_lens,
other_args=model_setup.extra_args,
)
if not success:
all_model_succeed = False
self.runner.add_report(results)
self.runner.write_final_report()
if not all_model_succeed:
raise AssertionError("Some models failed the perf tests.")
if __name__ == "__main__":
unittest.main()
+90
View File
@@ -0,0 +1,90 @@
import os
import unittest
import warnings
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.nightly_utils import NightlyBenchmarkRunner
from sglang.test.test_utils import (
DEFAULT_URL_FOR_TEST,
ModelLaunchSettings,
_parse_int_list_env,
parse_models,
)
register_cuda_ci(est_time=7200, suite="nightly-perf-vlm-2-gpu", nightly=True)
PROFILE_DIR = "performance_profiles_vlms"
MODEL_DEFAULTS = [
# Keep conservative defaults. Can be overridden by env NIGHTLY_VLM_MODELS
ModelLaunchSettings(
"Qwen/Qwen2.5-VL-7B-Instruct",
extra_args=["--mem-fraction-static=0.7"],
),
ModelLaunchSettings(
"google/gemma-3-27b-it",
),
ModelLaunchSettings("Qwen/Qwen3-VL-30B-A3B-Instruct", extra_args=["--tp=2"]),
# "OpenGVLab/InternVL2_5-2B",
# buggy in official transformers impl
# "openbmb/MiniCPM-V-2_6",
]
class TestNightlyVLMModelsPerformance(unittest.TestCase):
@classmethod
def setUpClass(cls):
warnings.filterwarnings(
"ignore", category=ResourceWarning, message="unclosed.*socket"
)
nightly_vlm_models_str = os.environ.get("NIGHTLY_VLM_MODELS")
if nightly_vlm_models_str:
cls.models = []
model_paths = parse_models(nightly_vlm_models_str)
for model_path in model_paths:
cls.models.append(ModelLaunchSettings(model_path))
else:
cls.models = MODEL_DEFAULTS
cls.base_url = DEFAULT_URL_FOR_TEST
cls.batch_sizes = _parse_int_list_env("NIGHTLY_VLM_BATCH_SIZES", "1,1,2,8,16")
cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_INPUT_LENS", "4096"))
cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_OUTPUT_LENS", "512"))
cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url)
cls.runner.setup_profile_directory()
def test_bench_one_batch(self):
all_model_succeed = True
for model_setup in self.models:
with self.subTest(model=model_setup.model_path):
# VLMs need additional benchmark args for dataset and trust-remote-code
extra_bench_args = [
"--trust-remote-code",
"--dataset-name=mmmu",
]
results, success, _ = self.runner.run_benchmark_for_model(
model_path=model_setup.model_path,
batch_sizes=self.batch_sizes,
input_lens=self.input_lens,
output_lens=self.output_lens,
other_args=model_setup.extra_args,
extra_bench_args=extra_bench_args,
)
if not success:
all_model_succeed = False
self.runner.add_report(results)
self.runner.write_final_report()
if not all_model_succeed:
raise AssertionError("Some models failed the perf tests.")
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,105 @@
import unittest
from types import SimpleNamespace
import requests
from sglang.srt.environ import envs
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
from sglang.test.test_utils import (
DEFAULT_DEEPSEEK_NVFP4_MODEL_FOR_TEST,
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
is_in_ci,
popen_launch_server,
write_github_step_summary,
)
# 16 GPU test (4 TP x 4 DP), runs on 2x 8-GPU B200 nodes
register_cuda_ci(est_time=600, suite="nightly-8-gpu-b200", nightly=True)
def test_gsm8k(base_url: str):
requests.get(base_url + "/flush_cache")
args = SimpleNamespace(
num_shots=5,
data_path=None,
num_questions=200,
max_new_tokens=512,
parallel=128,
host="http://127.0.0.1",
port=int(base_url.split(":")[-1]),
)
metrics = run_eval_few_shot_gsm8k(args)
server_info = requests.get(base_url + "/get_server_info")
avg_spec_accept_length = server_info.json()["internal_states"][0][
"avg_spec_accept_length"
]
print(f"{metrics=}")
print(f"{avg_spec_accept_length=}")
return metrics, avg_spec_accept_length
class TestEagleDPAttnServerLarge(CustomTestCase):
# FIXME: move this large mode test into nightly tests
@classmethod
def setUpClass(cls):
cls.model = DEFAULT_DEEPSEEK_NVFP4_MODEL_FOR_TEST
cls.base_url = DEFAULT_URL_FOR_TEST
other_args = [
"--tp-size",
"4",
"--dp-size",
"4",
"--enable-dp-attention",
"--attention-backend",
"trtllm_mla",
"--moe-runner-backend",
"flashinfer_trtllm",
"--quantization",
"modelopt_fp4",
"--speculative-algorithm",
"EAGLE",
"--speculative-num-steps",
"3",
"--speculative-eagle-topk",
"1",
"--speculative-num-draft-tokens",
"4",
"--kv-cache-dtype",
"fp8_e4m3",
"--model-loader-extra-config",
'{"enable_multithread_load": true,"num_threads": 64}',
]
with envs.SGLANG_ENABLE_SPEC_V2.override(True):
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
other_args=other_args,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_a_gsm8k(self):
metrics, avg_spec_accept_length = test_gsm8k(self.base_url)
self.assertGreater(metrics["accuracy"], 0.94)
# TODO: Update accept len to 2.04 once the bug is fixed
self.assertGreater(avg_spec_accept_length, 1.4)
if is_in_ci():
write_github_step_summary(
f"### test_gsm8k (deepseek-v3-fp4 mtp)\n"
f'{metrics["accuracy"]=:.3f}\n'
f"{avg_spec_accept_length=:.2f}\n"
)
if __name__ == "__main__":
unittest.main()
+276
View File
@@ -0,0 +1,276 @@
import argparse
import glob
import json
import os
import random
import subprocess
import sys
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
is_in_ci,
popen_launch_server,
)
register_cuda_ci(est_time=500, suite="nightly-4-gpu", nightly=True)
MODELS = [
SimpleNamespace(model="Qwen/Qwen2.5-VL-72B-Instruct", mmmu_accuracy=0.55),
SimpleNamespace(model="Qwen/Qwen3-VL-32B-Instruct", mmmu_accuracy=0.55),
SimpleNamespace(model="OpenGVLab/InternVL2_5-8B", mmmu_accuracy=0.52),
SimpleNamespace(model="zai-org/GLM-4.1V-9B-Thinking", mmmu_accuracy=0.68),
]
# Set default mem_fraction_static to 0.8
DEFAULT_MEM_FRACTION_STATIC = 0.8
class TestVLMEncoderDP(CustomTestCase):
parsed_args = None # Class variable to store args
@classmethod
def setUpClass(cls):
# Removed argument parsing from here
cls.base_url = DEFAULT_URL_FOR_TEST
cls.api_key = "sk-123456"
cls.time_out = DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
if cls.parsed_args is None:
cls.parsed_args = SimpleNamespace(
mem_fraction_static=DEFAULT_MEM_FRACTION_STATIC
)
# Set OpenAI API key and base URL environment variables. Needed for lmm-evals to work.
os.environ["OPENAI_API_KEY"] = cls.api_key
os.environ["OPENAI_API_BASE"] = f"{cls.base_url}/v1"
def run_mmmu_eval(
self,
model_version: str,
output_path: str,
*,
env: dict | None = None,
):
"""
Evaluate a VLM on the MMMU validation set with lmmseval.
Only `model_version` (checkpoint) and `chat_template` vary;
We are focusing only on the validation set due to resource constraints.
"""
# -------- fixed settings --------
model = "openai_compatible"
tp = 1
tasks = "mmmu_val"
batch_size = 32
log_suffix = "openai_compatible"
os.makedirs(output_path, exist_ok=True)
# -------- compose --model_args --------
model_args = f'model_version="{model_version}",' f"tp={tp}"
# -------- build command list --------
cmd = [
"python3",
"-m",
"lmms_eval",
"--model",
model,
"--model_args",
model_args,
"--tasks",
tasks,
"--batch_size",
str(batch_size),
"--log_samples",
"--log_samples_suffix",
log_suffix,
"--output_path",
str(output_path),
]
subprocess.run(
cmd,
check=True,
timeout=3600,
)
def _run_vlm_mmmu_test(
self,
model,
output_path,
test_name="",
custom_env=None,
log_level="info",
capture_output=False,
):
"""
Common method to run VLM MMMU benchmark test.
Args:
model: Model to test
output_path: Path for output logs
test_name: Optional test name for logging
custom_env: Optional custom environment variables
log_level: Log level for server (default: "info")
capture_output: Whether to capture server stdout/stderr
"""
print(f"\nTesting model: {model.model}{test_name}")
process = None
mmmu_accuracy = 0 # Initialize to handle potential exceptions
server_output = ""
try:
# Prepare environment variables
process_env = os.environ.copy()
if custom_env:
process_env.update(custom_env)
# if test vlm with cuda_ipc feature, open this env_var
process_env["SGLANG_USE_CUDA_IPC_TRANSPORT"] = "1"
# Prepare stdout/stderr redirection if needed
stdout_file = None
stderr_file = None
if capture_output:
stdout_file = open("/tmp/server_stdout.log", "w")
stderr_file = open("/tmp/server_stderr.log", "w")
# Launch server for testing
process = popen_launch_server(
model.model,
base_url=self.base_url,
timeout=self.time_out,
api_key=self.api_key,
other_args=[
"--trust-remote-code",
"--cuda-graph-max-bs",
"32",
"--mm-enable-dp-encoder",
"--tp=4",
"--mem-fraction-static",
str(self.parsed_args.mem_fraction_static), # Use class variable
"--log-level",
log_level,
],
env=process_env,
return_stdout_stderr=(
(stdout_file, stderr_file) if capture_output else None
),
)
# Run evaluation
self.run_mmmu_eval(model.model, output_path)
# Get the result file
# Search recursively for JSON result files (lmms-eval v0.4.1+ creates subdirectories)
result_files = glob.glob(f"{output_path}/**/*.json", recursive=True)
if not result_files:
result_files = glob.glob(f"{output_path}/*.json")
if not result_files:
raise FileNotFoundError(f"No JSON result files found in {output_path}")
result_file_path = result_files[0]
with open(result_file_path, "r") as f:
result = json.load(f)
print(f"Result{test_name}\n: {result}")
# Process the result
mmmu_accuracy = result["results"]["mmmu_val"]["mmmu_acc,none"]
print(
f"Model {model.model} achieved accuracy{test_name}: {mmmu_accuracy:.4f}"
)
# Capture server output if requested
if capture_output and process:
server_output = self._read_output_from_files()
# Assert performance meets expected threshold
self.assertGreaterEqual(
mmmu_accuracy,
model.mmmu_accuracy,
f"Model {model.model} accuracy ({mmmu_accuracy:.4f}) below expected threshold ({model.mmmu_accuracy:.4f}){test_name}",
)
return server_output
except Exception as e:
print(f"Error testing {model.model}{test_name}: {e}")
self.fail(f"Test failed for {model.model}{test_name}: {e}")
finally:
# Ensure process cleanup happens regardless of success/failure
if process is not None and process.poll() is None:
print(f"Cleaning up process {process.pid}")
try:
kill_process_tree(process.pid)
except Exception as e:
print(f"Error killing process: {e}")
# clean up temporary files
if capture_output:
if stdout_file:
stdout_file.close()
if stderr_file:
stderr_file.close()
for filename in ["/tmp/server_stdout.log", "/tmp/server_stderr.log"]:
try:
if os.path.exists(filename):
os.remove(filename)
except Exception as e:
print(f"Error removing {filename}: {e}")
def _read_output_from_files(self):
output_lines = []
log_files = [
("/tmp/server_stdout.log", "[STDOUT]"),
("/tmp/server_stderr.log", "[STDERR]"),
]
for filename, tag in log_files:
try:
if os.path.exists(filename):
with open(filename, "r") as f:
for line in f:
output_lines.append(f"{tag} {line.rstrip()}")
except Exception as e:
print(f"Error reading {tag.lower()} file: {e}")
return "\n".join(output_lines)
def test_vlm_mmmu_benchmark(self):
"""Test VLM models against MMMU benchmark."""
models_to_test = MODELS
if is_in_ci():
models_to_test = [random.choice(MODELS)]
for model in models_to_test:
self._run_vlm_mmmu_test(model, "./logs")
if __name__ == "__main__":
# Define and parse arguments here, before unittest.main
parser = argparse.ArgumentParser(description="Test VLM models")
parser.add_argument(
"--mem-fraction-static",
type=float,
help="Static memory fraction for the model",
default=DEFAULT_MEM_FRACTION_STATIC,
)
# Parse args intended for unittest
args = parser.parse_args()
# Store the parsed args object on the class
TestVLMEncoderDP.parsed_args = args
# Pass args to unittest
unittest.main(argv=[sys.argv[0]])