[mem_cache] Clean up unified allocator leftovers (#38103)
This commit is contained in:
@@ -545,6 +545,9 @@ class Envs:
|
||||
# Periodically log lazy-compaction stats per sub-pool (observability only).
|
||||
SGLANG_LOG_LAZY_COMPACTION_STATS = EnvBool(False)
|
||||
SGLANG_LOG_LAZY_COMPACTION_STATS_INTERVAL_SEC = EnvInt(30)
|
||||
# Per-call move cap on a non-urgent lazy-compaction flush, so a large
|
||||
# backlog cannot stall the scheduler loop; urgent flushes are uncapped.
|
||||
SGLANG_LAZY_COMPACTION_MAX_MOVES_PER_CALL = EnvInt(4096)
|
||||
# HND KV layout folds (page, head) into one paged index for per-kv-head sparse
|
||||
# page tables (DP attn); paged backends like trtllm_mha consume it directly.
|
||||
SGLANG_USE_HND_KVCACHE = EnvBool(False)
|
||||
|
||||
@@ -122,11 +122,9 @@ class BaseTokenToKVPoolAllocator(abc.ABC):
|
||||
return kv_indices
|
||||
|
||||
def get_cpu_copy(self, indices, mamba_indices=None):
|
||||
# FIXME: reuse the get_cpu_copy after paged allocator is implemented
|
||||
raise NotImplementedError()
|
||||
|
||||
def load_cpu_copy(self, kv_cache_cpu, indices, mamba_indices=None):
|
||||
# FIXME: reuse the load_cpu_copy after paged allocator is implemented
|
||||
raise NotImplementedError()
|
||||
|
||||
def alloc_extend(self, *args, **kwargs):
|
||||
|
||||
@@ -472,7 +472,8 @@ class DeepSeekV4HiSparseTokenToKVPoolAllocator(BaseTokenToKVPoolAllocator):
|
||||
self.hisparse_attn_allocator.free(buffer_indices[buffer_indices > 0])
|
||||
|
||||
def get_last_loc_compressed(self, last_locs: torch.Tensor):
|
||||
return (last_locs - 3) // self.compress_ratio
|
||||
# Last complete C4 block of a prefix of last_loc + 1 tokens; -1 stays -1.
|
||||
return (last_locs - (self.compress_ratio - 1)) // self.compress_ratio
|
||||
|
||||
def get_last_loc_hisparse_device(self, last_locs: torch.Tensor):
|
||||
return self.hisparse_kvcache._translate_loc_to_hisparse_device(
|
||||
|
||||
@@ -15,11 +15,6 @@ limitations under the License.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
"""
|
||||
Page-aligned memory pool.
|
||||
"""
|
||||
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import torch
|
||||
|
||||
@@ -878,6 +878,7 @@ class UnifiedMambaSWATokenToKVPoolAllocator(UnifiedSWATokenToKVPoolAllocator):
|
||||
at once, so re-check the JOINT gate instead of the per-side shortfall."""
|
||||
from sglang.srt.mem_cache.common import evict_from_tree_cache
|
||||
|
||||
# Arbitrary retry bound; a round that frees nothing ends the loop anyway.
|
||||
for _ in range(4):
|
||||
before = self.available_size()
|
||||
if before >= num_tokens:
|
||||
|
||||
@@ -306,11 +306,6 @@ class UnifiedMambaTokenToKVPoolAllocator(BaseTokenToKVPoolAllocator):
|
||||
self.full_attn_allocator.clear_inverse_history()
|
||||
self.mamba_allocator.clear_inverse_history()
|
||||
|
||||
def clear(self) -> None:
|
||||
self.full_attn_allocator.clear()
|
||||
self.mamba_allocator.clear()
|
||||
self.free_group = None
|
||||
|
||||
def free_segment(self, free_index: torch.Tensor, *, start_pos: int) -> None:
|
||||
"""Fixed-shape counterpart of `free()`; see `MultiEndedAllocator._page_reps`.
|
||||
The mamba sub-pool is slot-granular and untouched by a token free."""
|
||||
|
||||
@@ -361,8 +361,8 @@ class MultiEndedAllocator(BaseTokenToKVPoolAllocator):
|
||||
|
||||
# Per-call move cap on NON-urgent `_flush`: bounds work per `on_idle()` so
|
||||
# a large backlog doesn't block ZMQ IPC. Urgent retries are uncapped.
|
||||
self._lazy_max_moves_per_call = int(
|
||||
os.environ.get("SGLANG_LAZY_COMPACTION_MAX_MOVES_PER_CALL", "4096")
|
||||
self._lazy_max_moves_per_call = (
|
||||
envs.SGLANG_LAZY_COMPACTION_MAX_MOVES_PER_CALL.get()
|
||||
)
|
||||
|
||||
# Epoch-keyed memos for the capacity views: pure between mutations, but
|
||||
|
||||
Reference in New Issue
Block a user