[AMD] [Qwen 3.5 Day 0] Add Qwen 3.5 nightly accuracy tests (#19479)

This commit is contained in:
Michael
2026-03-02 19:42:42 -08:00
committed by GitHub
parent 060720c573
commit 6b8e62f94f
8 changed files with 367 additions and 1 deletions
@@ -0,0 +1,13 @@
model_name: "Qwen/Qwen3.5-397B-A17B"
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.9704
- name: "exact_match,flexible-extract"
value: 0.9697
limit: 1319
num_concurrent: 256
num_fewshot: 5
gen_kwargs: "max_gen_toks=2048"
rtol: 0.05
@@ -0,0 +1,64 @@
"""AMD Qwen 3.5 GSM8K lm-eval Evaluation Test (8-GPU)
Tests Qwen/Qwen3.5-397B-A17B (MoE, Hybrid Attention with Gated Delta Networks)
with lm-eval GSM8K benchmark on MI325/MI300X, matching the AMD Day 0 article.
Registry: nightly-amd-accuracy-8-gpu-qwen35 suite
"""
import os
import unittest
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_amd_ci
from sglang.test.kits.lm_eval_kit import LMEvalMixin
from sglang.test.test_utils import (
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
register_amd_ci(est_time=3600, suite="nightly-amd-accuracy-8-gpu-qwen35", nightly=True)
QWEN35_MODEL_PATH = "Qwen/Qwen3.5-397B-A17B"
SERVER_LAUNCH_TIMEOUT = 3600
TP_SIZE = 8
class TestQwen35EvalAMD(LMEvalMixin, CustomTestCase):
"""Qwen 3.5 GSM8K lm-eval Test for AMD MI325/MI300X."""
model_config_name = "lm_eval_configs/Qwen3.5-397B-A17B.yaml"
@classmethod
def setUpClass(cls):
cls.model = QWEN35_MODEL_PATH
cls.base_url = DEFAULT_URL_FOR_TEST
other_args = [
"--tp",
str(TP_SIZE),
"--attention-backend",
"triton",
"--trust-remote-code",
"--model-loader-extra-config",
'{"enable_multithread_load": true}',
"--watchdog-timeout",
"1200",
]
env = os.environ.copy()
env["SGLANG_USE_AITER"] = "1"
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=SERVER_LAUNCH_TIMEOUT,
other_args=other_args,
env=env,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,74 @@
"""MI35x Qwen 3.5 GSM8K lm-eval Evaluation Test (8-GPU)
Tests Qwen/Qwen3.5-397B-A17B (MoE, Hybrid Attention with Gated Delta Networks)
with lm-eval GSM8K benchmark on MI35x, matching the AMD Day 0 article.
Registry: nightly-amd-accuracy-8-gpu-mi35x-qwen35 suite
"""
import os
import unittest
import requests
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_amd_ci
from sglang.test.kits.lm_eval_kit import LMEvalMixin
from sglang.test.test_utils import (
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
register_amd_ci(
est_time=3600, suite="nightly-amd-accuracy-8-gpu-mi35x-qwen35", nightly=True
)
QWEN35_MODEL_PATH = "Qwen/Qwen3.5-397B-A17B"
SERVER_LAUNCH_TIMEOUT = 3600
TP_SIZE = 8
class TestQwen35EvalMI35x(LMEvalMixin, CustomTestCase):
"""Qwen 3.5 GSM8K lm-eval Test for AMD MI35x."""
model_config_name = "lm_eval_configs/Qwen3.5-397B-A17B.yaml"
@classmethod
def setUpClass(cls):
cls.model = QWEN35_MODEL_PATH
cls.base_url = DEFAULT_URL_FOR_TEST
def test_lm_eval(self):
"""Override to handle server lifecycle within test method (MI35x pattern)."""
other_args = [
"--tp",
str(TP_SIZE),
"--attention-backend",
"triton",
"--trust-remote-code",
"--model-loader-extra-config",
'{"enable_multithread_load": true}',
"--watchdog-timeout",
"1200",
]
env = os.environ.copy()
env["SGLANG_USE_AITER"] = "1"
process = popen_launch_server(
QWEN35_MODEL_PATH,
self.base_url,
timeout=SERVER_LAUNCH_TIMEOUT,
other_args=other_args,
env=env,
)
try:
requests.get(self.base_url + "/flush_cache")
super().test_lm_eval()
finally:
kill_process_tree(process.pid)
if __name__ == "__main__":
unittest.main()