diff --git a/.github/workflows/pr-test.yml b/.github/workflows/pr-test.yml index 21e18af8c..e46f7667b 100644 --- a/.github/workflows/pr-test.yml +++ b/.github/workflows/pr-test.yml @@ -1666,7 +1666,7 @@ jobs: strategy: fail-fast: false matrix: - part: [0, 1, 2] + part: [0, 1, 2, 3] steps: - name: Checkout code @@ -1695,7 +1695,7 @@ jobs: if [[ "${{ needs.check-changes.outputs.continue_on_error }}" == "true" ]]; then CONTINUE_ON_ERROR_FLAG="--continue-on-error" fi - IS_BLACKWELL=1 python3 run_suite.py --hw cuda --suite stage-c-test-4-gpu-b200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --timeout-per-file 1800 $CONTINUE_ON_ERROR_FLAG + IS_BLACKWELL=1 python3 run_suite.py --hw cuda --suite stage-c-test-4-gpu-b200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 4 --timeout-per-file 1800 $CONTINUE_ON_ERROR_FLAG - uses: ./.github/actions/upload-cuda-coredumps if: always() diff --git a/test/registered/4-gpu-models/test_qwen35_models.py b/test/registered/4-gpu-models/test_qwen35_models.py index a0fc77f3b..d67a298b7 100644 --- a/test/registered/4-gpu-models/test_qwen35_models.py +++ b/test/registered/4-gpu-models/test_qwen35_models.py @@ -1,3 +1,5 @@ +import shutil +import tempfile import unittest from types import SimpleNamespace @@ -17,13 +19,11 @@ from sglang.test.test_utils import ( popen_launch_server, ) -register_cuda_ci(est_time=1000, suite="stage-c-test-4-gpu-b200") +register_cuda_ci(est_time=1400, suite="stage-c-test-4-gpu-b200") QWEN35_FP4_MODEL = "nvidia/Qwen3.5-397B-A17B-NVFP4" - -ACC_THRESHOLDS = { - QWEN35_FP4_MODEL: {"gsm8k": 0.95}, -} +QWEN35_27B_MODEL = "Qwen/Qwen3.5-27B" +ACC_THRESHOLDS = {QWEN35_FP4_MODEL: {"gsm8k": 0.95}, QWEN35_27B_MODEL: {"gsm8k": 0.8}} class TestQwen35FP4(CustomTestCase): @@ -236,5 +236,78 @@ class TestQwen35FP4MTPV2(CustomTestCase): self.assertGreater(avg_spec_accept_length, 3.3) +class TestQwen35WithHiCache(CustomTestCase): + @classmethod + def setUpClass(cls): + cls.model = QWEN35_27B_MODEL + cls.base_url = DEFAULT_URL_FOR_TEST + cls.storage_dir = tempfile.mkdtemp(prefix="qwen35-hicache-") + env = { + "SGLANG_HICACHE_FILE_BACKEND_STORAGE_DIR": cls.storage_dir, + } + cls.process = popen_launch_server( + cls.model, + cls.base_url, + timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, + env=env, + other_args=[ + "--tp-size", + "4", + "--chunked-prefill-size", + "2048", + "--mamba-scheduler-strategy", + "extra_buffer", + "--mamba-track-interval", + "128", + "--mamba-ssm-dtype", + "bfloat16", + "--max-running-requests", + "128", + "--reasoning-parser", + "qwen3", + "--model-loader-extra-config", + '{"enable_multithread_load": true,"num_threads": 64}', + "--hicache-mem-layout", + "page_first_direct", + "--enable-hierarchical-cache", + "--hicache-ratio", + "2", + "--hicache-size", + "0", + "--hicache-write-policy", + "write_through", + "--hicache-storage-backend", + "file", + "--hicache-storage-prefetch-policy", + "wait_complete", + ], + ) + + @classmethod + def tearDownClass(cls): + kill_process_tree(cls.process.pid) + shutil.rmtree(cls.storage_dir, ignore_errors=True) + + def test_gsm8k(self): + args = SimpleNamespace( + model=self.model, + eval_name="gsm8k", + num_shots=5, + num_examples=200, + max_tokens=16000, + num_threads=128, + repeat=1, + temperature=0.6, + top_p=0.95, + top_k=20, + base_url=self.base_url, + host="http://127.0.0.1", + port=int(self.base_url.split(":")[-1]), + ) + metrics = run_eval(args) + print(f"{metrics=}") + self.assertGreaterEqual(metrics["score"], ACC_THRESHOLDS[self.model]["gsm8k"]) + + if __name__ == "__main__": unittest.main()