[DLLM] Add CI for diffusion LLMs (#14723)

This commit is contained in:
Zehuan Li
2025-12-19 08:54:03 +08:00
committed by GitHub
parent 9e0ef04e5b
commit b2803ff207
3 changed files with 87 additions and 3 deletions

View File

@@ -723,6 +723,7 @@ class Req:
self.dimensions = dimensions
# For diffusion LLM
self.dllm_ids = []
self.dllm_block_offset = 0
self.dllm_config = dllm_config
@@ -799,14 +800,15 @@ class Req:
return self.dllm_config is not None
def _init_fill_ids_for_dllm(self):
if not self.fill_ids:
self.fill_ids = (
if not self.dllm_ids:
self.dllm_ids = (
self.origin_input_ids
+ [self.dllm_config.mask_id] * self.dllm_config.block_size
)
else:
self.dllm_block_offset += self.dllm_config.block_size
self.fill_ids += [self.dllm_config.mask_id] * self.dllm_config.block_size
self.dllm_ids += [self.dllm_config.mask_id] * self.dllm_config.block_size
self.fill_ids = self.dllm_ids
def init_next_round_input(self, tree_cache: Optional[BasePrefixCache] = None):
if self.is_dllm():

View File

@@ -0,0 +1,81 @@
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
from sglang.test.send_one import BenchArgs, send_one_prompt
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
is_in_amd_ci,
is_in_ci,
popen_launch_server,
write_github_step_summary,
)
class TestLLaDA2Mini(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = "inclusionAI/LLaDA2.0-mini"
cls.base_url = DEFAULT_URL_FOR_TEST
other_args = [
"--trust-remote-code",
"--mem-fraction-static",
"0.9",
"--max-running-requests",
"1",
"--attention-backend",
"flashinfer",
"--dllm-algorithm",
"LowConfidence", # TODO: Add dLLM configurations
]
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
other_args=other_args,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_gsm8k(self):
args = SimpleNamespace(
num_shots=5,
data_path=None,
num_questions=200,
max_new_tokens=512,
parallel=128,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval_few_shot_gsm8k(args)
print(f"{metrics=}")
self.assertGreater(metrics["accuracy"], 0.6)
self.assertGreater(metrics["output_throughput"], 150)
def test_bs_1_speed(self):
args = BenchArgs(port=int(self.base_url.split(":")[-1]), max_new_tokens=2048)
acc_length, speed = send_one_prompt(args)
print(f"{speed=:.2f}")
if is_in_ci():
write_github_step_summary(
f"### test_bs_1_speed (llada2-mini) with tp1\n"
f"{speed=:.2f} token/s\n"
)
if is_in_amd_ci():
self.assertGreater(speed, 10)
else:
self.assertGreater(speed, 250)
if __name__ == "__main__":
unittest.main()

View File

@@ -51,6 +51,7 @@ suites = {
TestFile("rl/test_fp32_lm_head.py", 9),
# TestFile("rl/test_update_weights_from_disk.py", 210), # Temporarily disabled, see https://github.com/sgl-project/sglang/pull/13998
TestFile("rl/test_update_weights_from_tensor.py", 195),
TestFile("dllm/test_llada2_mini.py", 520),
TestFile("test_abort.py", 131),
TestFile("test_chunked_prefill.py", 312),
TestFile("test_create_kvindices.py", 7),