[AMD] Fall back to CPU tensor for decode retraction on ROCm (#36343)
This commit is contained in:
@@ -47,6 +47,7 @@ from sglang.srt.runtime_context import (
|
|||||||
get_parallel,
|
get_parallel,
|
||||||
get_schedule,
|
get_schedule,
|
||||||
)
|
)
|
||||||
|
from sglang.srt.utils import is_hip
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
|
|
||||||
@@ -143,6 +144,9 @@ def resolve_decode_retraction_backup(*, tp_worker: BaseTpWorker) -> str:
|
|||||||
backend = (
|
backend = (
|
||||||
"host_pool"
|
"host_pool"
|
||||||
if disagg.disaggregation_mode == "decode"
|
if disagg.disaggregation_mode == "decode"
|
||||||
|
# Large ROCm retraction restores can fault the GPU process. Keep
|
||||||
|
# host_pool opt-in on HIP until the retraction path is safe at scale.
|
||||||
|
and not is_hip()
|
||||||
and not get_parallel().dcp_enabled
|
and not get_parallel().dcp_enabled
|
||||||
and not disagg.disaggregation_decode_enable_radix_cache
|
and not disagg.disaggregation_decode_enable_radix_cache
|
||||||
# KV offload already owns a host pool; a second one double-books host memory.
|
# KV offload already owns a host pool; a second one double-books host memory.
|
||||||
|
|||||||
Reference in New Issue
Block a user