From 81e86992cdbd9dc35036b46741920aec6b80fa35 Mon Sep 17 00:00:00 2001 From: alisonshao <54658187+alisonshao@users.noreply.github.com> Date: Thu, 20 Nov 2025 18:00:02 -0800 Subject: [PATCH] [CI] Move nightly tests to test/nightly/ (#13683) --- .github/workflows/nightly-test-nvidia.yml | 32 ++--- test/{srt => }/nightly/nightly_utils.py | 0 .../test_batch_invariant_ops.py | 0 test/{srt => nightly}/test_cpp_radix_cache.py | 0 .../test_deepseek_r1_fp8_trtllm_backend.py | 0 test/nightly/test_deepseek_v31_perf.py | 87 ++++++++++++ .../test_deepseek_v32_nsabackend.py | 0 test/nightly/test_deepseek_v32_perf.py | 103 ++++++++++++++ .../test_deepseek_v3_deterministic.py | 0 .../test_deepseek_v3_fp4_cutlass_moe.py | 0 test/{srt => }/nightly/test_encoder_dp.py | 0 ...test_flashinfer_trtllm_gen_attn_backend.py | 0 .../test_flashinfer_trtllm_gen_moe_backend.py | 0 test/{srt => nightly}/test_fp4_moe.py | 0 .../nightly/test_gpt_oss_4gpu_perf.py | 0 .../test_lora_eviction_policy.py | 0 .../lora => nightly}/test_lora_openai_api.py | 0 .../test_lora_openai_compatible.py | 0 test/{srt/lora => nightly}/test_lora_qwen3.py | 0 .../lora => nightly}/test_lora_radix_cache.py | 0 .../nsa => nightly}/test_nsa_indexer.py | 0 .../test_qwen3_next_deterministic.py | 0 test/nightly/test_text_models_gsm8k_eval.py | 124 +++++++++++++++++ test/nightly/test_text_models_perf.py | 60 +++++++++ test/nightly/test_vlms_mmmu_eval.py | 127 ++++++++++++++++++ test/nightly/test_vlms_perf.py | 88 ++++++++++++ test/run_suite_nightly.py | 86 ++++++++++++ test/srt/run_suite.py | 32 +---- 28 files changed, 692 insertions(+), 47 deletions(-) rename test/{srt => }/nightly/nightly_utils.py (100%) rename test/{srt/batch_invariant => nightly}/test_batch_invariant_ops.py (100%) rename test/{srt => nightly}/test_cpp_radix_cache.py (100%) rename test/{srt => nightly}/test_deepseek_r1_fp8_trtllm_backend.py (100%) create mode 100644 test/nightly/test_deepseek_v31_perf.py rename test/{srt => nightly}/test_deepseek_v32_nsabackend.py (100%) create mode 100644 test/nightly/test_deepseek_v32_perf.py rename test/{srt => nightly}/test_deepseek_v3_deterministic.py (100%) rename test/{srt => nightly}/test_deepseek_v3_fp4_cutlass_moe.py (100%) rename test/{srt => }/nightly/test_encoder_dp.py (100%) rename test/{srt => }/nightly/test_flashinfer_trtllm_gen_attn_backend.py (100%) rename test/{srt => }/nightly/test_flashinfer_trtllm_gen_moe_backend.py (100%) rename test/{srt => nightly}/test_fp4_moe.py (100%) rename test/{srt => }/nightly/test_gpt_oss_4gpu_perf.py (100%) rename test/{srt/lora => nightly}/test_lora_eviction_policy.py (100%) rename test/{srt/lora => nightly}/test_lora_openai_api.py (100%) rename test/{srt/openai_server/features => nightly}/test_lora_openai_compatible.py (100%) rename test/{srt/lora => nightly}/test_lora_qwen3.py (100%) rename test/{srt/lora => nightly}/test_lora_radix_cache.py (100%) rename test/{srt/layers/attention/nsa => nightly}/test_nsa_indexer.py (100%) rename test/{srt => nightly}/test_qwen3_next_deterministic.py (100%) create mode 100644 test/nightly/test_text_models_gsm8k_eval.py create mode 100644 test/nightly/test_text_models_perf.py create mode 100644 test/nightly/test_vlms_mmmu_eval.py create mode 100644 test/nightly/test_vlms_perf.py create mode 100644 test/run_suite_nightly.py diff --git a/.github/workflows/nightly-test-nvidia.yml b/.github/workflows/nightly-test-nvidia.yml index 2ba48152e..26ef7c99e 100644 --- a/.github/workflows/nightly-test-nvidia.yml +++ b/.github/workflows/nightly-test-nvidia.yml @@ -30,8 +30,8 @@ jobs: - name: Run test timeout-minutes: 60 run: | - cd test/srt - python3 run_suite.py --suite nightly-1-gpu --continue-on-error + cd test + python3 run_suite_nightly.py --suite nightly-1-gpu --continue-on-error # General tests - 4 GPU H100 nightly-test-general-4-gpu-h100: @@ -48,8 +48,8 @@ jobs: - name: Run test timeout-minutes: 30 run: | - cd test/srt - python3 run_suite.py --suite nightly-4-gpu --continue-on-error + cd test + python3 run_suite_nightly.py --suite nightly-4-gpu --continue-on-error # General tests - 8 GPU H200 nightly-test-general-8-gpu-h200: @@ -70,8 +70,8 @@ jobs: env: GPU_CONFIG: "8-gpu-h200" run: | - cd test/srt - python3 run_suite.py --suite nightly-8-gpu-h200 --continue-on-error + cd test + python3 run_suite_nightly.py --suite nightly-8-gpu-h200 --continue-on-error # General tests - 8 GPU H20 nightly-test-general-8-gpu-h20: @@ -92,8 +92,8 @@ jobs: env: GPU_CONFIG: "8-gpu-h20" run: | - cd test/srt - python3 run_suite.py --suite nightly-8-gpu-h20 --continue-on-error + cd test + python3 run_suite_nightly.py --suite nightly-8-gpu-h20 --continue-on-error # Text model accuracy tests nightly-test-text-accuracy-2-gpu-runner: @@ -110,7 +110,7 @@ jobs: - name: Run eval test for text models timeout-minutes: 120 run: | - cd test/srt + cd test python3 nightly/test_text_models_gsm8k_eval.py # Text model performance tests @@ -132,7 +132,7 @@ jobs: PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "2-gpu-runner" run: | - cd test/srt + cd test rm -rf performance_profiles_text_models/ python3 nightly/test_text_models_perf.py @@ -159,7 +159,7 @@ jobs: - name: Run eval test for VLM models (fixed MMMU-100) timeout-minutes: 240 run: | - cd test/srt + cd test python3 nightly/test_vlms_mmmu_eval.py # VLM performance tests @@ -181,7 +181,7 @@ jobs: PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "2-gpu-runner" run: | - cd test/srt + cd test rm -rf performance_profiles_vlms/ python3 nightly/test_vlms_perf.py @@ -208,8 +208,8 @@ jobs: - name: Run test timeout-minutes: 60 run: | - cd test/srt - python3 run_suite.py --suite nightly-4-gpu-b200 --continue-on-error + cd test + python3 run_suite_nightly.py --suite nightly-4-gpu-b200 --continue-on-error # B200 Performance tests - 8 GPU nightly-test-perf-8-gpu-b200: @@ -233,7 +233,7 @@ jobs: GPU_CONFIG: "8-gpu-b200" run: | rm -rf test/srt/performance_profiles_deepseek_v31/ - cd test/srt + cd test IS_BLACKWELL=1 python3 nightly/test_deepseek_v31_perf.py - name: Publish DeepSeek v3.1 traces to storage repo @@ -252,7 +252,7 @@ jobs: GPU_CONFIG: "8-gpu-b200" run: | rm -rf test/srt/performance_profiles_deepseek_v32/ - cd test/srt + cd test IS_BLACKWELL=1 python3 nightly/test_deepseek_v32_perf.py - name: Publish DeepSeek v3.2 traces to storage repo diff --git a/test/srt/nightly/nightly_utils.py b/test/nightly/nightly_utils.py similarity index 100% rename from test/srt/nightly/nightly_utils.py rename to test/nightly/nightly_utils.py diff --git a/test/srt/batch_invariant/test_batch_invariant_ops.py b/test/nightly/test_batch_invariant_ops.py similarity index 100% rename from test/srt/batch_invariant/test_batch_invariant_ops.py rename to test/nightly/test_batch_invariant_ops.py diff --git a/test/srt/test_cpp_radix_cache.py b/test/nightly/test_cpp_radix_cache.py similarity index 100% rename from test/srt/test_cpp_radix_cache.py rename to test/nightly/test_cpp_radix_cache.py diff --git a/test/srt/test_deepseek_r1_fp8_trtllm_backend.py b/test/nightly/test_deepseek_r1_fp8_trtllm_backend.py similarity index 100% rename from test/srt/test_deepseek_r1_fp8_trtllm_backend.py rename to test/nightly/test_deepseek_r1_fp8_trtllm_backend.py diff --git a/test/nightly/test_deepseek_v31_perf.py b/test/nightly/test_deepseek_v31_perf.py new file mode 100644 index 000000000..58614350c --- /dev/null +++ b/test/nightly/test_deepseek_v31_perf.py @@ -0,0 +1,87 @@ +import unittest + +from nightly_utils import NightlyBenchmarkRunner + +from sglang.test.test_utils import DEFAULT_URL_FOR_TEST, _parse_int_list_env + +DEEPSEEK_V31_MODEL_PATH = "deepseek-ai/DeepSeek-V3.1" +PROFILE_DIR = "performance_profiles_deepseek_v31" + + +class TestNightlyDeepseekV31Performance(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.model = DEEPSEEK_V31_MODEL_PATH + cls.base_url = DEFAULT_URL_FOR_TEST + cls.batch_sizes = [1, 1, 8, 16, 64] + cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096")) + cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512")) + + # Define variant configurations + cls.variants = [ + { + "name": "basic", + "other_args": [ + "--trust-remote-code", + "--tp", + "8", + "--model-loader-extra-config", + '{"enable_multithread_load": true}', + ], + }, + { + "name": "mtp", + "other_args": [ + "--trust-remote-code", + "--tp", + "8", + "--speculative-algorithm", + "EAGLE", + "--speculative-num-steps", + "3", + "--speculative-eagle-topk", + "1", + "--speculative-num-draft-tokens", + "4", + "--mem-frac", + "0.7", + "--model-loader-extra-config", + '{"enable_multithread_load": true}', + ], + }, + ] + + cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) + cls.runner.setup_profile_directory() + + def test_bench_one_batch(self): + failed_variants = [] + + try: + for variant_config in self.variants: + with self.subTest(variant=variant_config["name"]): + results, success = self.runner.run_benchmark_for_model( + model_path=self.model, + batch_sizes=self.batch_sizes, + input_lens=self.input_lens, + output_lens=self.output_lens, + other_args=variant_config["other_args"], + variant=variant_config["name"], + ) + + if not success: + failed_variants.append(variant_config["name"]) + + self.runner.add_report(results) + finally: + self.runner.write_final_report() + + if failed_variants: + raise AssertionError( + f"Benchmark failed for {self.model} with the following variants: " + f"{', '.join(failed_variants)}" + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/srt/test_deepseek_v32_nsabackend.py b/test/nightly/test_deepseek_v32_nsabackend.py similarity index 100% rename from test/srt/test_deepseek_v32_nsabackend.py rename to test/nightly/test_deepseek_v32_nsabackend.py diff --git a/test/nightly/test_deepseek_v32_perf.py b/test/nightly/test_deepseek_v32_perf.py new file mode 100644 index 000000000..f7ed778c0 --- /dev/null +++ b/test/nightly/test_deepseek_v32_perf.py @@ -0,0 +1,103 @@ +import unittest + +from nightly_utils import NightlyBenchmarkRunner + +from sglang.test.test_utils import DEFAULT_URL_FOR_TEST, _parse_int_list_env + +DEEPSEEK_V32_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2-Exp" +PROFILE_DIR = "performance_profiles_deepseek_v32" + + +class TestNightlyDeepseekV32Performance(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.model = DEEPSEEK_V32_MODEL_PATH + cls.base_url = DEFAULT_URL_FOR_TEST + cls.batch_sizes = [1, 1, 8, 16, 64] + cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096")) + cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512")) + + # Define variant configurations + cls.variants = [ + { + "name": "basic", + "other_args": [ + "--trust-remote-code", + "--tp", + "8", + "--model-loader-extra-config", + '{"enable_multithread_load": true}', + ], + }, + { + "name": "mtp", + "other_args": [ + "--trust-remote-code", + "--tp", + "8", + "--speculative-algorithm", + "EAGLE", + "--speculative-num-steps", + "3", + "--speculative-eagle-topk", + "1", + "--speculative-num-draft-tokens", + "4", + "--mem-frac", + "0.7", + "--model-loader-extra-config", + '{"enable_multithread_load": true}', + ], + }, + { + "name": "nsa", + "other_args": [ + "--trust-remote-code", + "--tp", + "8", + "--attention-backend", + "nsa", + "--nsa-prefill-backend", + "flashmla_sparse", + "--nsa-decode-backend", + "flashmla_kv", + "--model-loader-extra-config", + '{"enable_multithread_load": true}', + ], + }, + ] + + cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) + cls.runner.setup_profile_directory() + + def test_bench_one_batch(self): + failed_variants = [] + + try: + for variant_config in self.variants: + with self.subTest(variant=variant_config["name"]): + results, success = self.runner.run_benchmark_for_model( + model_path=self.model, + batch_sizes=self.batch_sizes, + input_lens=self.input_lens, + output_lens=self.output_lens, + other_args=variant_config["other_args"], + variant=variant_config["name"], + ) + + if not success: + failed_variants.append(variant_config["name"]) + + self.runner.add_report(results) + finally: + self.runner.write_final_report() + + if failed_variants: + raise AssertionError( + f"Benchmark failed for {self.model} with the following variants: " + f"{', '.join(failed_variants)}" + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/srt/test_deepseek_v3_deterministic.py b/test/nightly/test_deepseek_v3_deterministic.py similarity index 100% rename from test/srt/test_deepseek_v3_deterministic.py rename to test/nightly/test_deepseek_v3_deterministic.py diff --git a/test/srt/test_deepseek_v3_fp4_cutlass_moe.py b/test/nightly/test_deepseek_v3_fp4_cutlass_moe.py similarity index 100% rename from test/srt/test_deepseek_v3_fp4_cutlass_moe.py rename to test/nightly/test_deepseek_v3_fp4_cutlass_moe.py diff --git a/test/srt/nightly/test_encoder_dp.py b/test/nightly/test_encoder_dp.py similarity index 100% rename from test/srt/nightly/test_encoder_dp.py rename to test/nightly/test_encoder_dp.py diff --git a/test/srt/nightly/test_flashinfer_trtllm_gen_attn_backend.py b/test/nightly/test_flashinfer_trtllm_gen_attn_backend.py similarity index 100% rename from test/srt/nightly/test_flashinfer_trtllm_gen_attn_backend.py rename to test/nightly/test_flashinfer_trtllm_gen_attn_backend.py diff --git a/test/srt/nightly/test_flashinfer_trtllm_gen_moe_backend.py b/test/nightly/test_flashinfer_trtllm_gen_moe_backend.py similarity index 100% rename from test/srt/nightly/test_flashinfer_trtllm_gen_moe_backend.py rename to test/nightly/test_flashinfer_trtllm_gen_moe_backend.py diff --git a/test/srt/test_fp4_moe.py b/test/nightly/test_fp4_moe.py similarity index 100% rename from test/srt/test_fp4_moe.py rename to test/nightly/test_fp4_moe.py diff --git a/test/srt/nightly/test_gpt_oss_4gpu_perf.py b/test/nightly/test_gpt_oss_4gpu_perf.py similarity index 100% rename from test/srt/nightly/test_gpt_oss_4gpu_perf.py rename to test/nightly/test_gpt_oss_4gpu_perf.py diff --git a/test/srt/lora/test_lora_eviction_policy.py b/test/nightly/test_lora_eviction_policy.py similarity index 100% rename from test/srt/lora/test_lora_eviction_policy.py rename to test/nightly/test_lora_eviction_policy.py diff --git a/test/srt/lora/test_lora_openai_api.py b/test/nightly/test_lora_openai_api.py similarity index 100% rename from test/srt/lora/test_lora_openai_api.py rename to test/nightly/test_lora_openai_api.py diff --git a/test/srt/openai_server/features/test_lora_openai_compatible.py b/test/nightly/test_lora_openai_compatible.py similarity index 100% rename from test/srt/openai_server/features/test_lora_openai_compatible.py rename to test/nightly/test_lora_openai_compatible.py diff --git a/test/srt/lora/test_lora_qwen3.py b/test/nightly/test_lora_qwen3.py similarity index 100% rename from test/srt/lora/test_lora_qwen3.py rename to test/nightly/test_lora_qwen3.py diff --git a/test/srt/lora/test_lora_radix_cache.py b/test/nightly/test_lora_radix_cache.py similarity index 100% rename from test/srt/lora/test_lora_radix_cache.py rename to test/nightly/test_lora_radix_cache.py diff --git a/test/srt/layers/attention/nsa/test_nsa_indexer.py b/test/nightly/test_nsa_indexer.py similarity index 100% rename from test/srt/layers/attention/nsa/test_nsa_indexer.py rename to test/nightly/test_nsa_indexer.py diff --git a/test/srt/test_qwen3_next_deterministic.py b/test/nightly/test_qwen3_next_deterministic.py similarity index 100% rename from test/srt/test_qwen3_next_deterministic.py rename to test/nightly/test_qwen3_next_deterministic.py diff --git a/test/nightly/test_text_models_gsm8k_eval.py b/test/nightly/test_text_models_gsm8k_eval.py new file mode 100644 index 000000000..8cd62e604 --- /dev/null +++ b/test/nightly/test_text_models_gsm8k_eval.py @@ -0,0 +1,124 @@ +import json +import unittest +import warnings +from types import SimpleNamespace + +from sglang.srt.utils import kill_process_tree +from sglang.test.run_eval import run_eval +from sglang.test.test_utils import ( + DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1, + DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2, + DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1, + DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2, + DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + DEFAULT_URL_FOR_TEST, + ModelLaunchSettings, + check_evaluation_test_results, + parse_models, + popen_launch_server, + write_results_to_json, +) + +MODEL_SCORE_THRESHOLDS = { + "meta-llama/Llama-3.1-8B-Instruct": 0.82, + "mistralai/Mistral-7B-Instruct-v0.3": 0.58, + "deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct": 0.85, + "google/gemma-2-27b-it": 0.91, + "meta-llama/Llama-3.1-70B-Instruct": 0.95, + "mistralai/Mixtral-8x7B-Instruct-v0.1": 0.616, + "Qwen/Qwen2-57B-A14B-Instruct": 0.86, + "neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8": 0.83, + "neuralmagic/Mistral-7B-Instruct-v0.3-FP8": 0.54, + "neuralmagic/DeepSeek-Coder-V2-Lite-Instruct-FP8": 0.835, + "zai-org/GLM-4.5-Air-FP8": 0.75, + # The threshold of neuralmagic/gemma-2-2b-it-FP8 should be 0.6, but this model has some accuracy regression. + # The fix is tracked at https://github.com/sgl-project/sglang/issues/4324, we set it to 0.50, for now, to make CI green. + "neuralmagic/gemma-2-2b-it-FP8": 0.50, + "neuralmagic/Meta-Llama-3.1-70B-Instruct-FP8": 0.94, + "neuralmagic/Mixtral-8x7B-Instruct-v0.1-FP8": 0.65, + "neuralmagic/Qwen2-72B-Instruct-FP8": 0.94, + "neuralmagic/Qwen2-57B-A14B-Instruct-FP8": 0.82, +} + + +# Do not use `CustomTestCase` since `test_mgsm_en_all_models` does not want retry +class TestNightlyGsm8KEval(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.models = [] + models_tp1 = parse_models( + DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1 + ) + parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1) + for model_path in models_tp1: + cls.models.append(ModelLaunchSettings(model_path, tp_size=1)) + + models_tp2 = parse_models( + DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2 + ) + parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2) + for model_path in models_tp2: + cls.models.append(ModelLaunchSettings(model_path, tp_size=2)) + + cls.base_url = DEFAULT_URL_FOR_TEST + + def test_mgsm_en_all_models(self): + warnings.filterwarnings( + "ignore", category=ResourceWarning, message="unclosed.*socket" + ) + is_first = True + all_results = [] + for model_setup in self.models: + with self.subTest(model=model_setup.model_path): + other_args = list(model_setup.extra_args) + + if model_setup.model_path == "meta-llama/Llama-3.1-70B-Instruct": + other_args.extend(["--mem-fraction-static", "0.9"]) + + process = popen_launch_server( + model=model_setup.model_path, + other_args=other_args, + base_url=self.base_url, + timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + ) + + try: + args = SimpleNamespace( + base_url=self.base_url, + model=model_setup.model_path, + eval_name="mgsm_en", + num_examples=None, + num_threads=1024, + ) + + metrics = run_eval(args) + print( + f"{'=' * 42}\n{model_setup.model_path} - metrics={metrics} score={metrics['score']}\n{'=' * 42}\n" + ) + + write_results_to_json( + model_setup.model_path, metrics, "w" if is_first else "a" + ) + is_first = False + + # 0.0 for empty latency + all_results.append((model_setup.model_path, metrics["score"], 0.0)) + finally: + kill_process_tree(process.pid) + + try: + with open("results.json", "r") as f: + print("\nFinal Results from results.json:") + print(json.dumps(json.load(f), indent=2)) + except Exception as e: + print(f"Error reading results.json: {e}") + + # Check all scores after collecting all results + check_evaluation_test_results( + all_results, + self.__class__.__name__, + model_accuracy_thresholds=MODEL_SCORE_THRESHOLDS, + model_count=len(self.models), + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/nightly/test_text_models_perf.py b/test/nightly/test_text_models_perf.py new file mode 100644 index 000000000..1e2cf70ff --- /dev/null +++ b/test/nightly/test_text_models_perf.py @@ -0,0 +1,60 @@ +import unittest + +from nightly_utils import NightlyBenchmarkRunner + +from sglang.test.test_utils import ( + DEFAULT_URL_FOR_TEST, + ModelLaunchSettings, + _parse_int_list_env, + parse_models, +) + +PROFILE_DIR = "performance_profiles_text_models" + + +class TestNightlyTextModelsPerformance(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.models = [] + # TODO: replace with DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1 or other model lists + for model_path in parse_models("meta-llama/Llama-3.1-8B-Instruct"): + cls.models.append(ModelLaunchSettings(model_path, tp_size=1)) + for model_path in parse_models("Qwen/Qwen2-57B-A14B-Instruct"): + cls.models.append(ModelLaunchSettings(model_path, tp_size=2)) + # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1), False, False), + # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2), False, True), + # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1), True, False), + # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2), True, True), + cls.base_url = DEFAULT_URL_FOR_TEST + cls.batch_sizes = [1, 1, 8, 16, 64] + cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096")) + cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512")) + cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) + cls.runner.setup_profile_directory() + + def test_bench_one_batch(self): + all_model_succeed = True + + for model_setup in self.models: + with self.subTest(model=model_setup.model_path): + results, success = self.runner.run_benchmark_for_model( + model_path=model_setup.model_path, + batch_sizes=self.batch_sizes, + input_lens=self.input_lens, + output_lens=self.output_lens, + other_args=model_setup.extra_args, + ) + + if not success: + all_model_succeed = False + + self.runner.add_report(results) + + self.runner.write_final_report() + + if not all_model_succeed: + raise AssertionError("Some models failed the perf tests.") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/nightly/test_vlms_mmmu_eval.py b/test/nightly/test_vlms_mmmu_eval.py new file mode 100644 index 000000000..aa2b43bd1 --- /dev/null +++ b/test/nightly/test_vlms_mmmu_eval.py @@ -0,0 +1,127 @@ +import json +import unittest +import warnings +from types import SimpleNamespace + +from sglang.srt.utils import kill_process_tree +from sglang.test.run_eval import run_eval +from sglang.test.test_utils import ( + DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + DEFAULT_URL_FOR_TEST, + ModelEvalMetrics, + ModelLaunchSettings, + check_evaluation_test_results, + popen_launch_server, + write_results_to_json, +) + +MODEL_THRESHOLDS = { + # Conservative thresholds on 100 MMMU samples, especially for latency thresholds + ModelLaunchSettings("deepseek-ai/deepseek-vl2-small"): ModelEvalMetrics( + 0.330, 56.1 + ), + ModelLaunchSettings("deepseek-ai/Janus-Pro-7B"): ModelEvalMetrics(0.285, 40.3), + ModelLaunchSettings("Efficient-Large-Model/NVILA-8B-hf"): ModelEvalMetrics( + 0.270, 56.7 + ), + ModelLaunchSettings("Efficient-Large-Model/NVILA-Lite-2B-hf"): ModelEvalMetrics( + 0.270, 23.8 + ), + ModelLaunchSettings("google/gemma-3-4b-it"): ModelEvalMetrics(0.360, 10.9), + ModelLaunchSettings("google/gemma-3n-E4B-it"): ModelEvalMetrics(0.360, 17.7), + ModelLaunchSettings("mistral-community/pixtral-12b"): ModelEvalMetrics(0.360, 16.6), + ModelLaunchSettings("moonshotai/Kimi-VL-A3B-Instruct"): ModelEvalMetrics( + 0.330, 22.3 + ), + ModelLaunchSettings("openbmb/MiniCPM-o-2_6"): ModelEvalMetrics(0.330, 29.3), + ModelLaunchSettings("openbmb/MiniCPM-v-2_6"): ModelEvalMetrics(0.259, 36.3), + ModelLaunchSettings("OpenGVLab/InternVL2_5-2B"): ModelEvalMetrics(0.300, 17.0), + ModelLaunchSettings("Qwen/Qwen2-VL-7B-Instruct"): ModelEvalMetrics(0.310, 83.3), + ModelLaunchSettings("Qwen/Qwen2.5-VL-7B-Instruct"): ModelEvalMetrics(0.340, 31.9), + ModelLaunchSettings( + "Qwen/Qwen3-VL-30B-A3B-Instruct", extra_args=["--tp=2"] + ): ModelEvalMetrics(0.29, 37.0), + ModelLaunchSettings( + "unsloth/Mistral-Small-3.1-24B-Instruct-2503" + ): ModelEvalMetrics(0.310, 16.7), + ModelLaunchSettings("XiaomiMiMo/MiMo-VL-7B-RL"): ModelEvalMetrics(0.28, 32.0), + ModelLaunchSettings("zai-org/GLM-4.1V-9B-Thinking"): ModelEvalMetrics(0.280, 30.4), +} + + +class TestNightlyVLMMmmuEval(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.models = list(MODEL_THRESHOLDS.keys()) + cls.base_url = DEFAULT_URL_FOR_TEST + + def test_mmmu_vlm_models(self): + warnings.filterwarnings( + "ignore", category=ResourceWarning, message="unclosed.*socket" + ) + is_first = True + all_results = [] + + for model in self.models: + model_path = model.model_path + with self.subTest(model=model_path): + process = popen_launch_server( + model=model_path, + base_url=self.base_url, + other_args=model.extra_args, + timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + ) + try: + args = SimpleNamespace( + base_url=self.base_url, + model=model_path, + eval_name="mmmu", + num_examples=100, + num_threads=64, + max_tokens=30, + ) + + args.return_latency = True + + metrics, latency = run_eval(args) + + metrics["score"] = round(metrics["score"], 4) + metrics["latency"] = round(latency, 4) + print( + f"{'=' * 42}\n{model_path} - metrics={metrics} score={metrics['score']}\n{'=' * 42}\n" + ) + + write_results_to_json(model_path, metrics, "w" if is_first else "a") + is_first = False + + all_results.append( + (model_path, metrics["score"], metrics["latency"]) + ) + finally: + kill_process_tree(process.pid) + + try: + with open("results.json", "r") as f: + print("\nFinal Results from results.json:") + print(json.dumps(json.load(f), indent=2)) + except Exception as e: + print(f"Error reading results: {e}") + + model_accuracy_thresholds = { + model.model_path: threshold.accuracy + for model, threshold in MODEL_THRESHOLDS.items() + } + model_latency_thresholds = { + model.model_path: threshold.eval_time + for model, threshold in MODEL_THRESHOLDS.items() + } + check_evaluation_test_results( + all_results, + self.__class__.__name__, + model_accuracy_thresholds=model_accuracy_thresholds, + model_latency_thresholds=model_latency_thresholds, + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/nightly/test_vlms_perf.py b/test/nightly/test_vlms_perf.py new file mode 100644 index 000000000..b837c0262 --- /dev/null +++ b/test/nightly/test_vlms_perf.py @@ -0,0 +1,88 @@ +import os +import unittest +import warnings + +from nightly_utils import NightlyBenchmarkRunner + +from sglang.test.test_utils import ( + DEFAULT_URL_FOR_TEST, + ModelLaunchSettings, + _parse_int_list_env, + parse_models, +) + +PROFILE_DIR = "performance_profiles_vlms" + +MODEL_DEFAULTS = [ + # Keep conservative defaults. Can be overridden by env NIGHTLY_VLM_MODELS + ModelLaunchSettings( + "Qwen/Qwen2.5-VL-7B-Instruct", + extra_args=["--mem-fraction-static=0.7"], + ), + ModelLaunchSettings( + "google/gemma-3-27b-it", + ), + ModelLaunchSettings("Qwen/Qwen3-VL-30B-A3B-Instruct", extra_args=["--tp=2"]), + # "OpenGVLab/InternVL2_5-2B", + # buggy in official transformers impl + # "openbmb/MiniCPM-V-2_6", +] + + +class TestNightlyVLMModelsPerformance(unittest.TestCase): + @classmethod + def setUpClass(cls): + warnings.filterwarnings( + "ignore", category=ResourceWarning, message="unclosed.*socket" + ) + + nightly_vlm_models_str = os.environ.get("NIGHTLY_VLM_MODELS") + if nightly_vlm_models_str: + cls.models = [] + model_paths = parse_models(nightly_vlm_models_str) + for model_path in model_paths: + cls.models.append(ModelLaunchSettings(model_path)) + else: + cls.models = MODEL_DEFAULTS + + cls.base_url = DEFAULT_URL_FOR_TEST + + cls.batch_sizes = _parse_int_list_env("NIGHTLY_VLM_BATCH_SIZES", "1,1,2,8,16") + cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_INPUT_LENS", "4096")) + cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_OUTPUT_LENS", "512")) + cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) + cls.runner.setup_profile_directory() + + def test_bench_one_batch(self): + all_model_succeed = True + + for model_setup in self.models: + with self.subTest(model=model_setup.model_path): + # VLMs need additional benchmark args for dataset and trust-remote-code + extra_bench_args = [ + "--trust-remote-code", + "--dataset-name=mmmu", + ] + + results, success = self.runner.run_benchmark_for_model( + model_path=model_setup.model_path, + batch_sizes=self.batch_sizes, + input_lens=self.input_lens, + output_lens=self.output_lens, + other_args=model_setup.extra_args, + extra_bench_args=extra_bench_args, + ) + + if not success: + all_model_succeed = False + + self.runner.add_report(results) + + self.runner.write_final_report() + + if not all_model_succeed: + raise AssertionError("Some models failed the perf tests.") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/run_suite_nightly.py b/test/run_suite_nightly.py new file mode 100644 index 000000000..43cee7c58 --- /dev/null +++ b/test/run_suite_nightly.py @@ -0,0 +1,86 @@ +import argparse +import os +from pathlib import Path + +from sglang.test.ci.ci_utils import TestFile, run_unittest_files + +# Nightly test suites +suites = { + "nightly-1-gpu": [ + TestFile("test_nsa_indexer.py", 2), + TestFile("test_lora_qwen3.py", 97), + TestFile("test_lora_radix_cache.py", 200), + TestFile("test_lora_eviction_policy.py", 200), + TestFile("test_lora_openai_api.py", 30), + TestFile("test_lora_openai_compatible.py", 150), + TestFile("test_batch_invariant_ops.py", 10), + TestFile("test_cpp_radix_cache.py", 60), + TestFile("test_deepseek_v3_deterministic.py", 240), + ], + "nightly-4-gpu-b200": [ + TestFile("test_flashinfer_trtllm_gen_moe_backend.py", 300), + TestFile("test_gpt_oss_4gpu_perf.py", 600), + TestFile("test_flashinfer_trtllm_gen_attn_backend.py", 300), + TestFile("test_deepseek_v3_fp4_cutlass_moe.py", 900), + TestFile("test_fp4_moe.py", 300), + ], + "nightly-8-gpu-b200": [ + TestFile("test_deepseek_r1_fp8_trtllm_backend.py", 3600), + ], + "nightly-4-gpu": [ + TestFile("test_encoder_dp.py", 500), + TestFile("test_qwen3_next_deterministic.py", 200), + ], + "nightly-8-gpu": [], + "nightly-8-gpu-h200": [ + TestFile("test_deepseek_v32_nsabackend.py", 600), + ], + "nightly-8-gpu-h20": [], +} + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument( + "--suite", + type=str, + required=True, + help="Test suite to run (e.g., nightly-1-gpu, nightly-4-gpu, etc.).", + ) + parser.add_argument( + "--timeout-per-file", + type=int, + default=1200, + help="The time limit for running one file in seconds (default: 1200).", + ) + parser.add_argument( + "--continue-on-error", + action="store_true", + default=False, + help="Continue running remaining tests even if one fails (default: False, useful for nightly tests).", + ) + args = parser.parse_args() + + if args.suite not in suites: + print(f"Error: Suite '{args.suite}' not found in available suites") + print(f"Available suites: {list(suites.keys())}") + exit(1) + + files = suites[args.suite] + + # Change directory to test/nightly where the test files are located + nightly_dir = Path(__file__).parent / "nightly" + os.chdir(nightly_dir) + + print(f"Running {len(files)} tests from suite: {args.suite}") + print(f"Test files: {[f.name for f in files]}") + + run_unittest_files( + files, + timeout_per_file=args.timeout_per_file, + continue_on_error=args.continue_on_error, + ) + + +if __name__ == "__main__": + main() diff --git a/test/srt/run_suite.py b/test/srt/run_suite.py index 964b06ec3..b5214dcd1 100644 --- a/test/srt/run_suite.py +++ b/test/srt/run_suite.py @@ -198,37 +198,7 @@ suites = { TestFile("test_quantization.py", 185), TestFile("test_gguf.py", 96), ], - # If the test cases take too long, considering adding them to nightly tests instead of per-commit tests - "nightly-1-gpu": [ - TestFile("layers/attention/nsa/test_nsa_indexer.py", 2), - TestFile("lora/test_lora_qwen3.py", 97), - TestFile("lora/test_lora_radix_cache.py", 200), - TestFile("lora/test_lora_eviction_policy.py", 200), - TestFile("lora/test_lora_openai_api.py", 30), - TestFile("openai_server/features/test_lora_openai_compatible.py", 150), - TestFile("batch_invariant/test_batch_invariant_ops.py", 10), - TestFile("test_cpp_radix_cache.py", 60), - TestFile("test_deepseek_v3_deterministic.py", 240), - ], - "nightly-4-gpu-b200": [ - TestFile("nightly/test_flashinfer_trtllm_gen_moe_backend.py", 300), - TestFile("nightly/test_gpt_oss_4gpu_perf.py", 600), - TestFile("nightly/test_flashinfer_trtllm_gen_attn_backend.py", 300), - TestFile("test_deepseek_v3_fp4_cutlass_moe.py", 900), - TestFile("test_fp4_moe.py", 300), - ], - "nightly-8-gpu-b200": [ - TestFile("test_deepseek_r1_fp8_trtllm_backend.py", 3600), - ], - "nightly-4-gpu": [ - TestFile("nightly/test_encoder_dp.py", 500), - TestFile("test_qwen3_next_deterministic.py", 200), - ], - "nightly-8-gpu": [], - "nightly-8-gpu-h200": [ - TestFile("test_deepseek_v32_nsabackend.py", 600), - ], - "nightly-8-gpu-h20": [], + # Nightly test suites have been moved to test/run_suite_nightly.py "__not_in_ci__": [ TestFile("test_bench_one_batch.py"), TestFile("test_bench_serving.py"),