From f9c04266922c6d3f07dd5b9968ffacda919c062f Mon Sep 17 00:00:00 2001 From: Bingxu Chen Date: Thu, 8 Jan 2026 14:43:48 +0800 Subject: [PATCH] [AMD CI] re-enable testcases missed when migrating ci test files (#16535) Co-authored-by: michael-amd Co-authored-by: yctseng0211 --- .github/workflows/pr-test-amd.yml | 12 ++++++------ test/registered/attention/test_create_kvindices.py | 3 ++- test/registered/attention/test_radix_attention.py | 3 ++- .../attention/test_torch_native_attention_backend.py | 3 ++- .../attention/test_triton_attention_backend.py | 3 ++- .../attention/test_triton_attention_kernels.py | 7 ++++++- .../attention/test_triton_sliding_window.py | 3 ++- test/registered/backends/test_torch_compile.py | 2 +- test/registered/lora/test_lora.py | 3 ++- test/registered/lora/test_lora_backend.py | 7 ++++++- test/registered/lora/test_lora_eviction.py | 3 ++- test/registered/lora/test_lora_tp.py | 7 ++++++- test/registered/lora/test_multi_lora_backend.py | 3 ++- test/registered/models/test_qwen_models.py | 2 +- test/registered/models/test_vlm_models.py | 2 +- .../openai_server/features/test_enable_thinking.py | 2 +- .../openai_server/features/test_json_mode.py | 2 +- test/registered/sampling/test_penalty.py | 2 +- test/srt/run_suite.py | 2 +- 19 files changed, 47 insertions(+), 24 deletions(-) diff --git a/.github/workflows/pr-test-amd.yml b/.github/workflows/pr-test-amd.yml index 63b6c1fd4..7573c6318 100644 --- a/.github/workflows/pr-test-amd.yml +++ b/.github/workflows/pr-test-amd.yml @@ -200,7 +200,7 @@ jobs: fail-fast: false matrix: runner: [linux-mi325-gpu-1] - part: [0, 1, 2, 3] + part: [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11] runs-on: ${{matrix.runner}} steps: - name: Checkout code @@ -222,7 +222,7 @@ jobs: - name: Run test timeout-minutes: 30 run: | - bash scripts/ci/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite stage-b-test-small-1-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 4 + bash scripts/ci/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite stage-b-test-small-1-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 12 stage-b-test-large-2-gpu-amd: needs: [check-changes, stage-a-test-1-amd] @@ -519,7 +519,7 @@ jobs: fail-fast: false matrix: runner: [linux-mi325-gpu-1] - part: [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11] + part: [0, 1] runs-on: ${{matrix.runner}} steps: - name: Checkout code @@ -541,7 +541,7 @@ jobs: - name: Run test timeout-minutes: 30 run: | - bash scripts/ci/amd_ci_exec.sh python3 run_suite.py --suite per-commit-amd --auto-partition-id ${{ matrix.part }} --auto-partition-size 12 + bash scripts/ci/amd_ci_exec.sh python3 run_suite.py --suite per-commit-amd --auto-partition-id ${{ matrix.part }} --auto-partition-size 2 unit-test-backend-1-gpu-amd-mi35x: needs: [check-changes, stage-a-test-1-amd] @@ -695,9 +695,9 @@ jobs: run: bash scripts/ci/amd_ci_install_dependency.sh - name: Run test - timeout-minutes: 30 + timeout-minutes: 60 run: | - bash scripts/ci/amd_ci_exec.sh python3 run_suite.py --suite per-commit-8-gpu-amd-mi35x --timeout-per-file 1800 + bash scripts/ci/amd_ci_exec.sh python3 run_suite.py --suite per-commit-8-gpu-amd-mi35x --timeout-per-file 3600 performance-test-1-gpu-part-1-amd: needs: [check-changes, stage-a-test-1-amd] diff --git a/test/registered/attention/test_create_kvindices.py b/test/registered/attention/test_create_kvindices.py index 0642aa29c..f5e9be464 100644 --- a/test/registered/attention/test_create_kvindices.py +++ b/test/registered/attention/test_create_kvindices.py @@ -4,11 +4,12 @@ import numpy as np import torch from sglang.srt.layers.attention.utils import create_flashinfer_kv_indices_triton -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.test_utils import CustomTestCase # Triton kernel unit test for KV indices creation register_cuda_ci(est_time=10, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=10, suite="stage-b-test-small-1-gpu") class TestCreateKvIndices(CustomTestCase): diff --git a/test/registered/attention/test_radix_attention.py b/test/registered/attention/test_radix_attention.py index c173d75bd..e72e4fa61 100644 --- a/test/registered/attention/test_radix_attention.py +++ b/test/registered/attention/test_radix_attention.py @@ -1,7 +1,7 @@ import unittest from sglang.srt.environ import envs -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.kits.radix_cache_server_kit import run_radix_attention_test from sglang.test.test_utils import ( DEFAULT_SMALL_MODEL_NAME_FOR_TEST, @@ -15,6 +15,7 @@ from sglang.test.test_utils import ( # RadixAttention server integration tests register_cuda_ci(est_time=100, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=100, suite="stage-b-test-small-1-gpu") class TestRadixCacheFCFS(CustomTestCase): diff --git a/test/registered/attention/test_torch_native_attention_backend.py b/test/registered/attention/test_torch_native_attention_backend.py index e6c6a9546..c7da08a03 100644 --- a/test/registered/attention/test_torch_native_attention_backend.py +++ b/test/registered/attention/test_torch_native_attention_backend.py @@ -7,7 +7,7 @@ import unittest from types import SimpleNamespace from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.run_eval import run_eval from sglang.test.test_utils import ( DEFAULT_MODEL_NAME_FOR_TEST, @@ -19,6 +19,7 @@ from sglang.test.test_utils import ( # Torch native attention backend integration test with MMLU eval register_cuda_ci(est_time=150, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=150, suite="stage-b-test-small-1-gpu") class TestTorchNativeAttnBackend(CustomTestCase): diff --git a/test/registered/attention/test_triton_attention_backend.py b/test/registered/attention/test_triton_attention_backend.py index d19bb6128..59cda72da 100644 --- a/test/registered/attention/test_triton_attention_backend.py +++ b/test/registered/attention/test_triton_attention_backend.py @@ -7,7 +7,7 @@ import unittest from types import SimpleNamespace from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.run_eval import run_eval from sglang.test.test_utils import ( DEFAULT_MODEL_NAME_FOR_TEST, @@ -21,6 +21,7 @@ from sglang.test.test_utils import ( # Triton attention backend integration test with latency benchmark and MMLU eval register_cuda_ci(est_time=200, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=1110, suite="stage-b-test-small-1-gpu") class TestTritonAttnBackend(CustomTestCase): diff --git a/test/registered/attention/test_triton_attention_kernels.py b/test/registered/attention/test_triton_attention_kernels.py index e2210ba2d..61c8a68ed 100644 --- a/test/registered/attention/test_triton_attention_kernels.py +++ b/test/registered/attention/test_triton_attention_kernels.py @@ -19,11 +19,16 @@ from sglang.srt.layers.attention.triton_ops.prefill_attention import ( context_attention_fwd, ) from sglang.srt.utils import get_device -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.test_utils import CustomTestCase # Triton attention kernel unit tests (decode, extend, prefill) register_cuda_ci(est_time=30, suite="stage-b-test-small-1-gpu") +register_amd_ci( + est_time=30, + suite="stage-b-test-small-1-gpu", + disabled="test was never enabled for AMD CI, needs validation", +) def extend_attention_fwd_torch( diff --git a/test/registered/attention/test_triton_sliding_window.py b/test/registered/attention/test_triton_sliding_window.py index 9c43b0cd5..439b220f0 100644 --- a/test/registered/attention/test_triton_sliding_window.py +++ b/test/registered/attention/test_triton_sliding_window.py @@ -4,7 +4,7 @@ from types import SimpleNamespace import requests from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.run_eval import run_eval from sglang.test.test_utils import ( DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, @@ -16,6 +16,7 @@ from sglang.test.test_utils import ( # Sliding window attention with Triton backend (Gemma-3 model) register_cuda_ci(est_time=100, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=100, suite="stage-b-test-small-1-gpu") class TestSlidingWindowAttentionTriton(CustomTestCase): diff --git a/test/registered/backends/test_torch_compile.py b/test/registered/backends/test_torch_compile.py index 0b3acda01..4588af347 100644 --- a/test/registered/backends/test_torch_compile.py +++ b/test/registered/backends/test_torch_compile.py @@ -17,7 +17,7 @@ from sglang.test.test_utils import ( ) register_cuda_ci(est_time=190, suite="stage-b-test-small-1-gpu") -register_amd_ci(est_time=993, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=1100, suite="stage-b-test-small-1-gpu") class TestTorchCompile(CustomTestCase): diff --git a/test/registered/lora/test_lora.py b/test/registered/lora/test_lora.py index 6d5c2b340..25ace89f0 100644 --- a/test/registered/lora/test_lora.py +++ b/test/registered/lora/test_lora.py @@ -16,7 +16,7 @@ import multiprocessing as mp import os import unittest -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.lora_utils import ( ALL_OTHER_MULTI_LORA_MODELS, CI_MULTI_LORA_MODELS, @@ -25,6 +25,7 @@ from sglang.test.lora_utils import ( from sglang.test.test_utils import CustomTestCase, is_in_ci register_cuda_ci(est_time=82, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=82, suite="stage-b-test-small-1-gpu") class TestLoRA(CustomTestCase): diff --git a/test/registered/lora/test_lora_backend.py b/test/registered/lora/test_lora_backend.py index e8f9134f6..0057c2a7f 100644 --- a/test/registered/lora/test_lora_backend.py +++ b/test/registered/lora/test_lora_backend.py @@ -17,7 +17,7 @@ import os import unittest from typing import List -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.lora_utils import ( ALL_OTHER_LORA_MODELS, BACKENDS, @@ -30,6 +30,11 @@ from sglang.test.lora_utils import ( from sglang.test.test_utils import CustomTestCase, is_in_ci register_cuda_ci(est_time=200, suite="stage-b-test-small-1-gpu") +register_amd_ci( + est_time=200, + suite="stage-b-test-small-1-gpu", + disabled="see https://github.com/sgl-project/sglang/issues/13107", +) class TestLoRABackend(CustomTestCase): diff --git a/test/registered/lora/test_lora_eviction.py b/test/registered/lora/test_lora_eviction.py index 7d9fb6f2e..05ed1dee3 100644 --- a/test/registered/lora/test_lora_eviction.py +++ b/test/registered/lora/test_lora_eviction.py @@ -19,11 +19,12 @@ from typing import Dict, List, Tuple import torch -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.runners import SRTRunner from sglang.test.test_utils import CustomTestCase register_cuda_ci(est_time=224, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=224, suite="stage-b-test-small-1-gpu") PROMPTS = [ "AI is a field of computer science focused on", diff --git a/test/registered/lora/test_lora_tp.py b/test/registered/lora/test_lora_tp.py index a8cddf6d6..a23b8300a 100644 --- a/test/registered/lora/test_lora_tp.py +++ b/test/registered/lora/test_lora_tp.py @@ -17,7 +17,7 @@ import os import unittest from typing import List -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.lora_utils import ( ALL_OTHER_LORA_MODELS, CI_LORA_MODELS, @@ -29,6 +29,11 @@ from sglang.test.lora_utils import ( from sglang.test.test_utils import CustomTestCase, is_in_ci register_cuda_ci(est_time=116, suite="stage-b-test-large-2-gpu") +register_amd_ci( + est_time=116, + suite="stage-b-test-large-2-gpu-amd", + disabled="see https://github.com/sgl-project/sglang/issues/13107", +) class TestLoRATP(CustomTestCase): diff --git a/test/registered/lora/test_multi_lora_backend.py b/test/registered/lora/test_multi_lora_backend.py index 58dfdc16c..1a43a13f2 100644 --- a/test/registered/lora/test_multi_lora_backend.py +++ b/test/registered/lora/test_multi_lora_backend.py @@ -16,7 +16,7 @@ import multiprocessing as mp import os import unittest -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.lora_utils import ( ALL_OTHER_MULTI_LORA_MODELS, CI_MULTI_LORA_MODELS, @@ -25,6 +25,7 @@ from sglang.test.lora_utils import ( from sglang.test.test_utils import CustomTestCase, is_in_ci register_cuda_ci(est_time=60, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=60, suite="stage-b-test-small-1-gpu") # All prompts are used at once in a batch. PROMPTS = [ diff --git a/test/registered/models/test_qwen_models.py b/test/registered/models/test_qwen_models.py index 026bb9ef2..beba81b7f 100644 --- a/test/registered/models/test_qwen_models.py +++ b/test/registered/models/test_qwen_models.py @@ -2,7 +2,7 @@ from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci # Qwen model tests register_cuda_ci(est_time=90, suite="stage-b-test-small-1-gpu") -register_amd_ci(est_time=82, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=130, suite="stage-b-test-small-1-gpu") import unittest from types import SimpleNamespace diff --git a/test/registered/models/test_vlm_models.py b/test/registered/models/test_vlm_models.py index 58b6a7a01..4afd39a00 100644 --- a/test/registered/models/test_vlm_models.py +++ b/test/registered/models/test_vlm_models.py @@ -2,7 +2,7 @@ from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci # VLM (Vision Language Model) tests register_cuda_ci(est_time=270, suite="stage-b-test-small-1-gpu") -register_amd_ci(est_time=387, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=420, suite="stage-b-test-small-1-gpu") import argparse import random diff --git a/test/registered/openai_server/features/test_enable_thinking.py b/test/registered/openai_server/features/test_enable_thinking.py index 17e71d99b..d9e4ae83a 100644 --- a/test/registered/openai_server/features/test_enable_thinking.py +++ b/test/registered/openai_server/features/test_enable_thinking.py @@ -22,7 +22,7 @@ from sglang.test.test_utils import ( ) register_cuda_ci(est_time=70, suite="stage-b-test-small-1-gpu") -register_amd_ci(est_time=70, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=200, suite="stage-b-test-small-1-gpu") class TestEnableThinking(CustomTestCase): diff --git a/test/registered/openai_server/features/test_json_mode.py b/test/registered/openai_server/features/test_json_mode.py index b646be287..38395d1c9 100644 --- a/test/registered/openai_server/features/test_json_mode.py +++ b/test/registered/openai_server/features/test_json_mode.py @@ -14,7 +14,7 @@ from sglang.test.test_utils import ( ) register_cuda_ci(est_time=109, suite="stage-b-test-small-1-gpu") -register_amd_ci(est_time=120, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=180, suite="stage-b-test-small-1-gpu") class TestJSONModeMixin: diff --git a/test/registered/sampling/test_penalty.py b/test/registered/sampling/test_penalty.py index a074dfaee..921d34cc1 100644 --- a/test/registered/sampling/test_penalty.py +++ b/test/registered/sampling/test_penalty.py @@ -10,7 +10,7 @@ from sglang.srt.utils import kill_process_tree from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci register_cuda_ci(est_time=82, suite="stage-b-test-small-1-gpu") -register_amd_ci(est_time=180, suite="stage-b-test-small-1-gpu") +register_amd_ci(est_time=82, suite="stage-b-test-small-1-gpu") from sglang.test.test_utils import ( DEFAULT_SMALL_MODEL_NAME_FOR_TEST, DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, diff --git a/test/srt/run_suite.py b/test/srt/run_suite.py index 7b73c4098..958c939b9 100644 --- a/test/srt/run_suite.py +++ b/test/srt/run_suite.py @@ -113,7 +113,7 @@ suite_amd = { TestFile("test_deepseek_v3_mtp.py", 275), ], "per-commit-8-gpu-amd-mi35x": [ - TestFile("test_deepseek_r1_mxfp4_8gpu.py", 1800), + TestFile("test_deepseek_r1_mxfp4_8gpu.py", 3600), ], "nightly-amd": [ TestFile("nightly/test_gsm8k_eval_amd.py"),