Add tool call tests for DeepSeek V3.2 in nightly CI (#17951)
This commit is contained in:
@@ -14,6 +14,11 @@ from sglang.test.performance_test_runner import (
|
||||
run_performance_test,
|
||||
)
|
||||
from sglang.test.test_utils import DEFAULT_URL_FOR_TEST, ModelLaunchSettings, is_in_ci
|
||||
from sglang.test.tool_call_test_runner import (
|
||||
ToolCallTestParams,
|
||||
ToolCallTestResult,
|
||||
run_tool_call_test,
|
||||
)
|
||||
|
||||
|
||||
def run_combined_tests(
|
||||
@@ -23,8 +28,9 @@ def run_combined_tests(
|
||||
is_vlm: bool = False,
|
||||
accuracy_params: Optional[AccuracyTestParams] = None,
|
||||
performance_params: Optional[PerformanceTestParams] = None,
|
||||
tool_call_params: Optional[ToolCallTestParams] = None,
|
||||
) -> dict:
|
||||
"""Run performance and/or accuracy tests for a list of models.
|
||||
"""Run performance, accuracy, and/or tool call tests for a list of models.
|
||||
|
||||
Args:
|
||||
models: List of ModelLaunchSettings to test
|
||||
@@ -33,6 +39,7 @@ def run_combined_tests(
|
||||
is_vlm: Whether these are VLM models (affects defaults)
|
||||
accuracy_params: Parameters for accuracy tests (None to skip accuracy)
|
||||
performance_params: Parameters for performance tests (None to skip perf)
|
||||
tool_call_params: Parameters for tool call tests (None to skip tool call)
|
||||
|
||||
Returns:
|
||||
dict with test results:
|
||||
@@ -52,6 +59,7 @@ def run_combined_tests(
|
||||
base_url = base_url or DEFAULT_URL_FOR_TEST
|
||||
run_perf = performance_params is not None
|
||||
run_accuracy = accuracy_params is not None
|
||||
run_tool_call = tool_call_params is not None
|
||||
|
||||
# Print test header
|
||||
print("\n" + "=" * 80)
|
||||
@@ -61,6 +69,8 @@ def run_combined_tests(
|
||||
print(f" Accuracy dataset: {accuracy_params.dataset}")
|
||||
if run_perf:
|
||||
print(f" Performance batches: {performance_params.batch_sizes}")
|
||||
if run_tool_call:
|
||||
print(" Tool call tests: enabled")
|
||||
print("=" * 80)
|
||||
|
||||
# Set up performance parameters
|
||||
@@ -96,6 +106,7 @@ def run_combined_tests(
|
||||
"model": model.model_path,
|
||||
"perf_result": None,
|
||||
"accuracy_result": None,
|
||||
"tool_call_result": None,
|
||||
"errors": [],
|
||||
}
|
||||
|
||||
@@ -136,6 +147,21 @@ def run_combined_tests(
|
||||
print("\nWaiting 20 seconds for resource cleanup...")
|
||||
time.sleep(20)
|
||||
|
||||
# Run tool call test
|
||||
if run_tool_call:
|
||||
tc_result: ToolCallTestResult = run_tool_call_test(
|
||||
model=model,
|
||||
params=tool_call_params,
|
||||
base_url=base_url,
|
||||
)
|
||||
model_result["tool_call_result"] = tc_result
|
||||
if not tc_result.passed:
|
||||
all_passed = False
|
||||
model_result["errors"].extend(tc_result.failures)
|
||||
|
||||
print("\nWaiting 20 seconds for resource cleanup...")
|
||||
time.sleep(20)
|
||||
|
||||
all_results.append(model_result)
|
||||
|
||||
# Write performance report if we ran perf tests
|
||||
@@ -180,6 +206,11 @@ def run_combined_tests(
|
||||
print(f" Accuracy: {'PASS' if acc.passed else 'FAIL'}")
|
||||
if acc.score is not None:
|
||||
print(f" Score: {acc.score:.3f}")
|
||||
if run_tool_call and model_result["tool_call_result"]:
|
||||
tc = model_result["tool_call_result"]
|
||||
print(
|
||||
f" Tool Call: {'PASS' if tc.passed else 'FAIL'} ({tc.num_passed}/{tc.num_total})"
|
||||
)
|
||||
if model_result["errors"]:
|
||||
print(f" Errors: {model_result['errors']}")
|
||||
|
||||
|
||||
Reference in New Issue
Block a user