[NPU][1/N] NPU basic functions refactor and new modelslim quant type (#13359)
This commit is contained in:
@@ -30,13 +30,7 @@ from sglang.srt.model_loader.weight_utils import (
|
||||
from sglang.srt.models.qwen2 import Qwen2MLP as Qwen3MLP
|
||||
from sglang.srt.models.qwen2 import Qwen2Model
|
||||
from sglang.srt.server_args import get_global_server_args
|
||||
from sglang.srt.utils import (
|
||||
add_prefix,
|
||||
get_cmo_stream,
|
||||
is_cuda,
|
||||
is_npu,
|
||||
wait_cmo_stream,
|
||||
)
|
||||
from sglang.srt.utils import add_prefix, is_cuda, is_npu
|
||||
|
||||
Qwen3Config = None
|
||||
|
||||
@@ -47,6 +41,8 @@ _is_npu = is_npu()
|
||||
if _is_npu:
|
||||
from sgl_kernel_npu.norm.split_qkv_rmsnorm_rope import split_qkv_rmsnorm_rope
|
||||
|
||||
from sglang.srt.hardware_backend.npu.cmo import get_cmo_stream, wait_cmo_stream
|
||||
|
||||
|
||||
class Qwen3Attention(nn.Module):
|
||||
def __init__(
|
||||
|
||||
Reference in New Issue
Block a user