From 834795adb827385c8b2630be93f6e9e2fa5e9da9 Mon Sep 17 00:00:00 2001 From: Yuwei An Date: Mon, 9 Mar 2026 23:49:25 -0700 Subject: [PATCH] [CI] Refactor PCG related CI (#19994) Signed-off-by: yuweia Signed-off-by: Oasis-Git --- .../test_piecewise_cuda_graph_2_gpu.py | 60 ---- .../test_piecewise_cuda_graph_large_1_gpu.py | 131 -------- .../test_piecewise_cuda_graph_small_1_gpu.py | 294 ------------------ ...test_piecewise_cuda_graph_support_1_gpu.py | 135 ++++++++ 4 files changed, 135 insertions(+), 485 deletions(-) delete mode 100644 test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_2_gpu.py delete mode 100644 test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_large_1_gpu.py delete mode 100644 test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_small_1_gpu.py create mode 100644 test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_support_1_gpu.py diff --git a/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_2_gpu.py b/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_2_gpu.py deleted file mode 100644 index 324e1eae1..000000000 --- a/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_2_gpu.py +++ /dev/null @@ -1,60 +0,0 @@ -import unittest - -from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci -from sglang.test.run_eval import run_eval -from sglang.test.test_utils import ( - DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - DEFAULT_URL_FOR_TEST, - CustomTestCase, - SimpleNamespace, - popen_launch_server, -) - -# CI Registration - 2-GPU tests (80GB GPUs required) -register_cuda_ci(est_time=160, suite="stage-b-test-large-2-gpu") - - -class TestPiecewiseCudaGraphTP(CustomTestCase): - """Test piecewise CUDA graph with normal TP""" - - @classmethod - def setUpClass(cls): - cls.model = "Qwen/Qwen3-Coder-30B-A3B-Instruct" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[ - "--piecewise-cuda-graph-compiler", - "eager", - "--tp", - "2", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_gsm8k_accuracy(self): - """Test GSM8K accuracy with 8-shot setting""" - num_examples = 2000 - - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - - metrics = run_eval(args) - print(f"GSM8K Accuracy: {metrics['score']:.3f}") - - self.assertGreaterEqual(metrics["score"], 0.90) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_large_1_gpu.py b/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_large_1_gpu.py deleted file mode 100644 index 6fa32efd4..000000000 --- a/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_large_1_gpu.py +++ /dev/null @@ -1,131 +0,0 @@ -import unittest - -from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci -from sglang.test.run_eval import run_eval -from sglang.test.test_utils import ( - DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - DEFAULT_URL_FOR_TEST, - CustomTestCase, - SimpleNamespace, - popen_launch_server, -) - -# CI Registration - Large 1-GPU tests (80GB GPU required) -register_cuda_ci(est_time=480, suite="stage-b-test-large-1-gpu") - - -class TestPiecewiseCudaGraphQwen3MoE(CustomTestCase): - """Test piecewise CUDA graph with Qwen3-Coder-30B-A3B-Instruct MoE model""" - - @classmethod - def setUpClass(cls): - cls.model = "Qwen/Qwen3-Coder-30B-A3B-Instruct" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[ - "--piecewise-cuda-graph-compiler", - "eager", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_gsm8k_accuracy(self): - """Test GSM8K accuracy with 8-shot setting""" - num_examples = 2000 - - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - - metrics = run_eval(args) - print(f"GSM8K Accuracy: {metrics['score']:.3f}") - - self.assertGreaterEqual(metrics["score"], 0.90) - - -class TestPiecewiseCudaGraphGPTQ(CustomTestCase): - - @classmethod - def setUpClass(cls): - cls.model = "Qwen/Qwen3-30B-A3B-GPTQ-Int4" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_mgsm_accuracy(self): - num_examples = 1319 - - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - - metrics = run_eval(args) - print(f"MGSM Accuracy: {metrics['score']:.3f}") - - # Expected accuracy: 0.948, allow some variance - self.assertGreaterEqual(metrics["score"], 0.92) - - -class TestPiecewiseCudaGraphAWQ(CustomTestCase): - """Test piecewise CUDA graph with AWQ quantized model""" - - @classmethod - def setUpClass(cls): - cls.model = "Qwen/QwQ-32B-AWQ" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_mgsm_accuracy(self): - """Test MGSM accuracy with AWQ model""" - num_examples = 1319 - - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - - metrics = run_eval(args) - print(f"MGSM Accuracy: {metrics['score']:.3f}") - print(f"Output throughput: {metrics.get('throughput', 'N/A')} token/s") - - # Expected accuracy: 0.680, allow some variance - self.assertGreaterEqual(metrics["score"], 0.65) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_small_1_gpu.py b/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_small_1_gpu.py deleted file mode 100644 index 60c14d37a..000000000 --- a/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_small_1_gpu.py +++ /dev/null @@ -1,294 +0,0 @@ -import unittest - -import torch - -from sglang import Engine -from sglang.lang.chat_template import get_chat_template_by_model_path -from sglang.srt.utils import get_device_sm, kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci -from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k -from sglang.test.run_eval import run_eval -from sglang.test.test_utils import ( - DEFAULT_IMAGE_URL, - DEFAULT_MODEL_NAME_FOR_TEST, - DEFAULT_MODEL_NAME_FOR_TEST_MLA, - DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - DEFAULT_URL_FOR_TEST, - CustomTestCase, - SimpleNamespace, - popen_launch_server, - run_bench_one_batch, -) - -# CI Registration - Small 1-GPU tests (24GB GPU sufficient) -register_cuda_ci(est_time=539, suite="stage-b-test-large-1-gpu") - - -class TestPiecewiseCudaGraphCorrectness(CustomTestCase): - @classmethod - def setUpClass(cls): - cls.model = DEFAULT_MODEL_NAME_FOR_TEST - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_mmlu(self): - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mmlu", - num_examples=64, - num_threads=32, - ) - - metrics = run_eval(args) - self.assertGreaterEqual(metrics["score"], 0.65) - - -class TestPiecewiseCudaGraphBenchmark(CustomTestCase): - - def test_latency(self): - prefill_latency, _, _ = run_bench_one_batch( - DEFAULT_MODEL_NAME_FOR_TEST, other_args=[] - ) - self.assertLess(prefill_latency, 0.015) - - -@unittest.skipIf(get_device_sm() < 100, "Test requires CUDA SM 100 or higher") -class TestPiecewiseCudaGraphLlama31FP4(CustomTestCase): - """MGSM test: piecewise CUDA graph with NVFP4 Llama3.1 8B on Blackwell.""" - - @classmethod - def setUpClass(cls): - cls.model = "nvidia/Llama-3.1-8B-Instruct-FP4" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[ - "--quantization", - "modelopt_fp4", - "--mem-fraction-static", - "0.8", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_mgsm_accuracy(self): - num_examples = 1319 - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - metrics = run_eval(args) - print(f"MGSM Accuracy: {metrics['score']:.3f}") - self.assertGreaterEqual(metrics["score"], 0.78) - - -class TestPiecewiseCudaGraphDeepSeek(CustomTestCase): - @classmethod - def setUpClass(cls): - cls.model = DEFAULT_MODEL_NAME_FOR_TEST_MLA - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[ - "--piecewise-cuda-graph-compiler", - "eager", - "--piecewise-cuda-graph-max-tokens", - "4096", # should less than max_context_len - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_gsm8k(self): - args = SimpleNamespace( - num_shots=5, - data_path=None, - num_questions=200, - max_new_tokens=512, - parallel=128, - host="http://127.0.0.1", - port=int(self.base_url.split(":")[-1]), - ) - metrics = run_eval_few_shot_gsm8k(args) - print(metrics) - - self.assertGreater(metrics["accuracy"], 0.62) - - -class TestPiecewiseCudaGraphFP8(CustomTestCase): - """Test piecewise CUDA graph with FP8 quantized model""" - - @classmethod - def setUpClass(cls): - cls.model = "nvidia/Llama-3.1-8B-Instruct-FP8" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[ - "--quantization", - "modelopt_fp8", - "--kv-cache-dtype", - "bfloat16", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_mgsm_accuracy(self): - """Test MGSM accuracy with FP8 model""" - num_examples = 1319 - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - metrics = run_eval(args) - self.assertGreaterEqual(metrics["score"], 0.85) - print(f"MGSM Accuracy: {metrics['score']:.3f}") - - -class TestPiecewiseCudaGraphQwen25VL(CustomTestCase): - """Test piecewise CUDA graph with Qwen2.5-VL-7B-Instruct model""" - - @classmethod - def setUpClass(cls): - cls.model = "Qwen/Qwen2.5-VL-7B-Instruct" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[ - "--piecewise-cuda-graph-compiler", - "eager", - "--disable-radix-cache", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_gsm8k_accuracy(self): - """Test GSM8K accuracy with 8-shot setting""" - num_examples = 2000 - - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - - metrics = run_eval(args) - print(f"GSM8K Accuracy: {metrics['score']:.3f}") - - self.assertGreaterEqual(metrics["score"], 0.70) - - -class TestPiecewiseCudaGraphInternVL25(CustomTestCase): - """Test piecewise CUDA graph with InternVL2.5-8B-Instruct model""" - - @classmethod - def setUpClass(cls): - cls.model = "OpenGVLab/InternVL2_5-8B" - cls.base_url = DEFAULT_URL_FOR_TEST - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - other_args=[ - "--piecewise-cuda-graph-compiler", - "eager", - "--disable-radix-cache", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def test_gsm8k_accuracy(self): - """Test GSM8K accuracy with 8-shot setting""" - num_examples = 2000 - - args = SimpleNamespace( - base_url=self.base_url, - model=self.model, - eval_name="mgsm_en", - num_examples=num_examples, - num_threads=min(num_examples, 1024), - ) - - metrics = run_eval(args) - print(f"GSM8K Accuracy: {metrics['score']:.3f}") - - self.assertGreaterEqual(metrics["score"], 0.70) - - -class TestPiecewiseCudaGraphQwen25VLEmbedding(CustomTestCase): - """Test piecewise CUDA graph with Qwen2.5-VL-3B-Instruct embedding model""" - - def test_embedding(self): - model_path = "Qwen/Qwen2.5-VL-3B-Instruct" - chat_template = get_chat_template_by_model_path(model_path) - text = f"{chat_template.image_token}What is in this picture? Answer: " - - engine = Engine( - model_path=model_path, - enable_multimodal=True, - is_embedding=True, - piecewise_cuda_graph_compiler="eager", - ) - out = engine.encode([text], image_data=[DEFAULT_IMAGE_URL])[0]["embedding"] - engine.shutdown() - self.assertGreater(len(out), 0) - - engine = Engine( - model_path=model_path, - enable_multimodal=True, - is_embedding=True, - disable_piecewise_cuda_graph=True, - ) - out_without_pcg = engine.encode([text], image_data=[DEFAULT_IMAGE_URL])[0][ - "embedding" - ] - engine.shutdown() - self.assertGreater(len(out_without_pcg), 0) - - self.assertTrue( - torch.allclose(torch.tensor(out), torch.tensor(out_without_pcg)) - ) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_support_1_gpu.py b/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_support_1_gpu.py new file mode 100644 index 000000000..14a4abeaa --- /dev/null +++ b/test/registered/piecewise_cuda_graph/test_piecewise_cuda_graph_support_1_gpu.py @@ -0,0 +1,135 @@ +import unittest + +import torch + +from sglang import Engine +from sglang.lang.chat_template import get_chat_template_by_model_path +from sglang.srt.utils import kill_process_tree +from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.run_eval import run_eval +from sglang.test.test_utils import ( + DEFAULT_IMAGE_URL, + DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + DEFAULT_URL_FOR_TEST, + CustomTestCase, + SimpleNamespace, + popen_launch_server, +) + +# CI Registration +register_cuda_ci(est_time=220, suite="stage-b-test-large-1-gpu") + + +class TestPiecewiseCudaGraphQwen25VL(CustomTestCase): + """Test piecewise CUDA graph with Qwen2.5-VL-7B-Instruct model""" + + @classmethod + def setUpClass(cls): + cls.model = "Qwen/Qwen2.5-VL-7B-Instruct" + cls.base_url = DEFAULT_URL_FOR_TEST + cls.process = popen_launch_server( + cls.model, + cls.base_url, + timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + other_args=[ + "--enforce-piecewise-cuda-graph", + "--disable-radix-cache", + ], + ) + + @classmethod + def tearDownClass(cls): + kill_process_tree(cls.process.pid) + + def test_mgsm_accuracy(self): + num_examples = 2000 + + args = SimpleNamespace( + base_url=self.base_url, + model=self.model, + eval_name="mgsm_en", + num_examples=num_examples, + num_threads=min(num_examples, 1024), + ) + + metrics = run_eval(args) + print(f"MGSM Accuracy: {metrics['score']:.3f}") + + self.assertGreaterEqual(metrics["score"], 0.70) + + +class TestPiecewiseCudaGraphInternVL25(CustomTestCase): + """Test piecewise CUDA graph with InternVL2.5-8B model""" + + @classmethod + def setUpClass(cls): + cls.model = "OpenGVLab/InternVL2_5-8B" + cls.base_url = DEFAULT_URL_FOR_TEST + cls.process = popen_launch_server( + cls.model, + cls.base_url, + timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + other_args=[ + "--enforce-piecewise-cuda-graph", + "--disable-radix-cache", + ], + ) + + @classmethod + def tearDownClass(cls): + kill_process_tree(cls.process.pid) + + def test_mgsm_accuracy(self): + num_examples = 2000 + + args = SimpleNamespace( + base_url=self.base_url, + model=self.model, + eval_name="mgsm_en", + num_examples=num_examples, + num_threads=min(num_examples, 1024), + ) + + metrics = run_eval(args) + print(f"MGSM Accuracy: {metrics['score']:.3f}") + + self.assertGreaterEqual(metrics["score"], 0.70) + + +class TestPiecewiseCudaGraphQwen25VLEmbedding(CustomTestCase): + """Test piecewise CUDA graph with Qwen2.5-VL-3B-Instruct embedding model""" + + def test_embedding(self): + model_path = "Qwen/Qwen2.5-VL-3B-Instruct" + chat_template = get_chat_template_by_model_path(model_path) + text = f"{chat_template.image_token}What is in this picture? Answer: " + + engine = Engine( + model_path=model_path, + enable_multimodal=True, + is_embedding=True, + enforce_piecewise_cuda_graph=True, + ) + out = engine.encode([text], image_data=[DEFAULT_IMAGE_URL])[0]["embedding"] + engine.shutdown() + self.assertGreater(len(out), 0) + + engine = Engine( + model_path=model_path, + enable_multimodal=True, + is_embedding=True, + disable_piecewise_cuda_graph=True, + ) + out_without_pcg = engine.encode([text], image_data=[DEFAULT_IMAGE_URL])[0][ + "embedding" + ] + engine.shutdown() + self.assertGreater(len(out_without_pcg), 0) + + self.assertTrue( + torch.allclose(torch.tensor(out), torch.tensor(out_without_pcg)) + ) + + +if __name__ == "__main__": + unittest.main()