[NPU][1/N] NPU basic functions refactor and new modelslim quant type (#13359)

This commit is contained in:
Even Zhou
2025-12-04 16:15:31 +08:00
committed by GitHub
parent d6c490192d
commit 894c0dc57c
43 changed files with 2500 additions and 2058 deletions
+3 -7
View File
@@ -30,13 +30,7 @@ from sglang.srt.model_loader.weight_utils import (
from sglang.srt.models.qwen2 import Qwen2MLP as Qwen3MLP
from sglang.srt.models.qwen2 import Qwen2Model
from sglang.srt.server_args import get_global_server_args
from sglang.srt.utils import (
add_prefix,
get_cmo_stream,
is_cuda,
is_npu,
wait_cmo_stream,
)
from sglang.srt.utils import add_prefix, is_cuda, is_npu
Qwen3Config = None
@@ -47,6 +41,8 @@ _is_npu = is_npu()
if _is_npu:
from sgl_kernel_npu.norm.split_qkv_rmsnorm_rope import split_qkv_rmsnorm_rope
from sglang.srt.hardware_backend.npu.cmo import get_cmo_stream, wait_cmo_stream
class Qwen3Attention(nn.Module):
def __init__(