diff --git a/python/sglang/srt/layers/attention/nsa/nsa_indexer.py b/python/sglang/srt/layers/attention/nsa/nsa_indexer.py index ebc22ea71..f5db3d7a3 100644 --- a/python/sglang/srt/layers/attention/nsa/nsa_indexer.py +++ b/python/sglang/srt/layers/attention/nsa/nsa_indexer.py @@ -169,10 +169,9 @@ class Indexer(CustomOp): @torch.compile(dynamic=True) def _get_logits_head_gate(self, x: torch.Tensor, q_scale: torch.Tensor): - # Keep x in original dtype (bfloat16) for projection, then convert to float32 - weights, _ = self.weights_proj(x) - weights = weights.float() * self.n_heads**-0.5 - weights = weights.unsqueeze(-1) * q_scale.float() * self.softmax_scale + weights, _ = self.weights_proj(x.float()) + weights = weights * self.n_heads**-0.5 + weights = weights.unsqueeze(-1) * q_scale * self.softmax_scale return weights def _get_q_k_bf16( diff --git a/python/sglang/srt/mem_cache/radix_cache_cpp.py b/python/sglang/srt/mem_cache/radix_cache_cpp.py index 8bb08351d..e8f187b46 100644 --- a/python/sglang/srt/mem_cache/radix_cache_cpp.py +++ b/python/sglang/srt/mem_cache/radix_cache_cpp.py @@ -45,8 +45,8 @@ class RadixCacheCpp(BasePrefixCache): self.write_through_threshold = ( 1 if server_args.hicache_write_policy == "write_through" else 2 ) - self.token_to_kv_pool_allocator = params.token_to_kv_pool_allocator self.device = self.token_to_kv_pool_allocator.device + self.token_to_kv_pool_allocator = params.token_to_kv_pool_allocator self.req_to_token_pool = params.req_to_token_pool self.page_size = params.page_size self.kv_cache = self.token_to_kv_pool_allocator.get_kvcache()