diff --git a/python/sglang/srt/model_executor/model_runner.py b/python/sglang/srt/model_executor/model_runner.py index 2e4497cd5..0660c9687 100644 --- a/python/sglang/srt/model_executor/model_runner.py +++ b/python/sglang/srt/model_executor/model_runner.py @@ -2365,8 +2365,9 @@ class ModelRunner: backend_str = self.server_args.moe_runner_backend if backend_str not in [ "flashinfer_trtllm", - "flashinfer_cutlass", "flashinfer_mxfp4", + # TODO: flashinfer_cutlass will cause some flashinfer compilation errors. To be fixed. + # "flashinfer_cutlass", ]: return False diff --git a/test/nightly/test_deepseek_v3_fp4_cutlass_moe.py b/test/nightly/test_deepseek_v3_fp4_cutlass_moe.py deleted file mode 100644 index c3a509efa..000000000 --- a/test/nightly/test_deepseek_v3_fp4_cutlass_moe.py +++ /dev/null @@ -1,75 +0,0 @@ -import unittest -from types import SimpleNamespace - -from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci -from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k -from sglang.test.test_utils import ( - DEFAULT_URL_FOR_TEST, - CustomTestCase, - is_in_ci, - popen_launch_server, - write_github_step_summary, -) - -register_cuda_ci(est_time=900, suite="nightly-4-gpu-b200", nightly=True) - -FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4" -SERVER_LAUNCH_TIMEOUT = 1000 - - -class TestDeepseekV3FP4CutlassMoE(CustomTestCase): - @classmethod - def setUpClass(cls): - cls.model = FULL_DEEPSEEK_V3_FP4_MODEL_PATH - cls.base_url = DEFAULT_URL_FOR_TEST - other_args = [ - "--tp", - "4", - "--ep", - "4", - "--attention-backend", - "trtllm_mla", - "--moe-runner-backend", - "flashinfer_cutlass", - "--quantization", - "modelopt_fp4", - "--model-loader-extra-config", - '{"enable_multithread_load": true}', - ] - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=SERVER_LAUNCH_TIMEOUT, - other_args=other_args, - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_a_gsm8k( - self, - ): # Append an "a" to make this test run first (alphabetically) to warm up the server - args = SimpleNamespace( - num_shots=8, - data_path=None, - num_questions=1319, - parallel=1319, - max_new_tokens=512, - host="http://127.0.0.1", - port=int(self.base_url.split(":")[-1]), - ) - metrics = run_eval_few_shot_gsm8k(args) - print(f"{metrics=}") - - if is_in_ci(): - write_github_step_summary( - f"### test_gsm8k (deepseek-v3-fp4-cutlass-moe)\n" - f'{metrics["accuracy"]=:.3f}\n' - ) - self.assertGreater(metrics["accuracy"], 0.935) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/run_suite_nightly.py b/test/run_suite_nightly.py index 676d598cd..6e6c701b0 100644 --- a/test/run_suite_nightly.py +++ b/test/run_suite_nightly.py @@ -23,7 +23,6 @@ suites = { TestFile("test_flashinfer_trtllm_gen_moe_backend.py", 300), TestFile("test_gpt_oss_4gpu_perf.py", 600), TestFile("test_flashinfer_trtllm_gen_attn_backend.py", 300), - TestFile("test_deepseek_v3_fp4_cutlass_moe.py", 900), TestFile("test_fp4_moe.py", 300), TestFile("test_qwen3_fp4_trtllm_gen_moe.py", 300), TestFile("test_eagle_infer_beta_dp_attention_large.py", 600), diff --git a/test/srt/test_deepseek_v3_fp4_4gpu.py b/test/srt/test_deepseek_v3_fp4_4gpu.py index e173cb6f1..faf291fa5 100644 --- a/test/srt/test_deepseek_v3_fp4_4gpu.py +++ b/test/srt/test_deepseek_v3_fp4_4gpu.py @@ -1,3 +1,4 @@ +import os import unittest from types import SimpleNamespace @@ -149,5 +150,62 @@ class TestDeepseekV3FP4PiecewiseCudaGraph(CustomTestCase): self.assertGreater(speed, 120) +class TestDeepseekV3FP4CutlassMoE(CustomTestCase): + @classmethod + def setUpClass(cls): + cls.model = FULL_DEEPSEEK_V3_FP4_MODEL_PATH + cls.base_url = DEFAULT_URL_FOR_TEST + other_args = [ + "--tp", + "4", + "--ep", + "4", + "--attention-backend", + "trtllm_mla", + "--moe-runner-backend", + "flashinfer_cutlass", + "--quantization", + "modelopt_fp4", + "--model-loader-extra-config", + '{"enable_multithread_load": true}', + ] + cls.process = popen_launch_server( + cls.model, + cls.base_url, + timeout=SERVER_LAUNCH_TIMEOUT, + other_args=other_args, + env={ + **os.environ, + "SGLANG_MOE_NVFP4_DISPATCH": "1", # Enable nvfp4 all gather + }, + ) + + @classmethod + def tearDownClass(cls): + kill_process_tree(cls.process.pid) + + def test_a_gsm8k( + self, + ): # Append an "a" to make this test run first (alphabetically) to warm up the server + args = SimpleNamespace( + num_shots=8, + data_path=None, + num_questions=1319, + parallel=1319, + max_new_tokens=512, + host="http://127.0.0.1", + port=int(self.base_url.split(":")[-1]), + ) + metrics = run_eval_few_shot_gsm8k(args) + print(f"{metrics=}") + + if is_in_ci(): + write_github_step_summary( + f"### test_gsm8k (deepseek-v3-fp4-cutlass-moe)\n" + f'{metrics["accuracy"]=:.3f}\n' + ) + self.assertGreater(metrics["accuracy"], 0.935) + + if __name__ == "__main__": unittest.main()