Fix LMCache unit test and init bug (#14005)
Signed-off-by: DongDongJu <commisori28@gmail.com>
This commit is contained in:
@@ -20,6 +20,7 @@ except ImportError as e:
|
|||||||
"LMCache is not installed. Please install it by running `pip install lmcache`"
|
"LMCache is not installed. Please install it by running `pip install lmcache`"
|
||||||
) from e
|
) from e
|
||||||
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from sglang.srt.configs.model_config import ModelConfig
|
from sglang.srt.configs.model_config import ModelConfig
|
||||||
from sglang.srt.managers.schedule_batch import Req
|
from sglang.srt.managers.schedule_batch import Req
|
||||||
@@ -211,7 +212,18 @@ class LMCRadixCache(RadixCache):
|
|||||||
if not is_insert:
|
if not is_insert:
|
||||||
return
|
return
|
||||||
|
|
||||||
kv_committed_len = req.pop_committed_kv_cache()
|
from sglang.srt.server_args import get_global_server_args
|
||||||
|
|
||||||
|
global_server_args = get_global_server_args()
|
||||||
|
topk = global_server_args.speculative_eagle_topk
|
||||||
|
enable_kv_committed_len = topk is None or topk == 1
|
||||||
|
if enable_kv_committed_len:
|
||||||
|
kv_committed_len = req.kv_committed_len
|
||||||
|
else:
|
||||||
|
kv_committed_len = len(req.origin_input_ids) + max(
|
||||||
|
len(req.output_ids) - 1, 0
|
||||||
|
)
|
||||||
|
|
||||||
token_ids = (req.origin_input_ids + req.output_ids)[:kv_committed_len]
|
token_ids = (req.origin_input_ids + req.output_ids)[:kv_committed_len]
|
||||||
kv_indices = self.req_to_token_pool.req_to_token[
|
kv_indices = self.req_to_token_pool.req_to_token[
|
||||||
req.req_pool_idx, :kv_committed_len
|
req.req_pool_idx, :kv_committed_len
|
||||||
@@ -254,27 +266,3 @@ class LMCRadixCache(RadixCache):
|
|||||||
)
|
)
|
||||||
except Exception: # pragma: no cover
|
except Exception: # pragma: no cover
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
from sglang.srt.mem_cache.cache_init_params import CacheInitParams
|
|
||||||
|
|
||||||
params = CacheInitParams(
|
|
||||||
req_to_token_pool=None,
|
|
||||||
token_to_kv_pool_allocator=None,
|
|
||||||
page_size=1,
|
|
||||||
disable=False,
|
|
||||||
enable_kv_cache_events=False,
|
|
||||||
)
|
|
||||||
cache = LMCRadixCache(
|
|
||||||
params=params,
|
|
||||||
model_config=None,
|
|
||||||
tp_size=1,
|
|
||||||
rank=0,
|
|
||||||
tp_group=None,
|
|
||||||
)
|
|
||||||
cache.insert(RadixKey([1, 2, 3]), torch.tensor([10, 11, 12], dtype=torch.int64))
|
|
||||||
cache.insert(
|
|
||||||
RadixKey([1, 2, 3, 4]), torch.tensor([10, 11, 12, 13], dtype=torch.int64)
|
|
||||||
)
|
|
||||||
cache.pretty_print()
|
|
||||||
|
|||||||
Reference in New Issue
Block a user