Support Expert Deferral Mechanism in KTransformers (#12586)

Co-authored-by: Chen Hongtao <56470055+chenht2022@users.noreply.github.com>
Co-authored-by: chenht2022 <cht22@mails.tsinghua.edu.cn>
This commit is contained in:
Atream
2025-11-05 13:41:52 -08:00
committed by GitHub
co-authored by Chen Hongtao chenht2022
parent 3651cfbf62
commit 627bac649c
3 changed files with 50 additions and 1 deletions
+2
View File
@@ -270,6 +270,8 @@ class Envs:
SGLANG_KT_MOE_AMX_WEIGHT_PATH = EnvStr(None)
SGLANG_KT_AMX_METHOD = EnvStr(None)
SGLANG_KT_MOE_CHUNKED_PREFILL_SIZE = EnvInt(None)
SGLANG_KT_MOE_MAX_DEFERRED_EXPERTS_PER_TOKEN = EnvInt(None)
SGLANG_KT_MOE_TOTAL_LAYERS = EnvInt(None)
# Sparse Embeddings
SGLANG_EMBEDDINGS_SPARSE_HEAD = EnvStr(None)