[PD] Support KV transfer with MORI-IO (#14626)
Co-authored-by: cwortman-amd <cwortman@amd.com>
This commit is contained in:
@@ -59,6 +59,16 @@ ARG TILELANG_COMMIT="ebf4a7cb8881432165ae8760e99d209d905c704a"
|
|||||||
ARG FHT_REPO="https://github.com/jeffdaily/fast-hadamard-transform.git"
|
ARG FHT_REPO="https://github.com/jeffdaily/fast-hadamard-transform.git"
|
||||||
ARG FHT_BRANCH="rocm"
|
ARG FHT_BRANCH="rocm"
|
||||||
ARG FHT_COMMIT="46efb7d776d38638fc39f3c803eaee3dd7016bd1"
|
ARG FHT_COMMIT="46efb7d776d38638fc39f3c803eaee3dd7016bd1"
|
||||||
|
|
||||||
|
ARG ENABLE_MORI=0
|
||||||
|
ARG NIC_BACKEND=none
|
||||||
|
|
||||||
|
ARG MORI_REPO="https://github.com/ROCm/mori.git"
|
||||||
|
ARG MORI_COMMIT="b0dce4beebeb1f26c784eee17d5fd9785ee9447f"
|
||||||
|
|
||||||
|
# AMD AINIC apt repo settings
|
||||||
|
ARG AINIC_VERSION=1.117.5
|
||||||
|
ARG UBUNTU_CODENAME=jammy
|
||||||
USER root
|
USER root
|
||||||
|
|
||||||
# Install some basic utilities
|
# Install some basic utilities
|
||||||
@@ -283,6 +293,73 @@ RUN python3 -m pip install --no-cache-dir \
|
|||||||
py-spy \
|
py-spy \
|
||||||
pre-commit
|
pre-commit
|
||||||
|
|
||||||
|
# -----------------------
|
||||||
|
# MORI (optional)
|
||||||
|
RUN /bin/bash -lc 'set -euo pipefail; \
|
||||||
|
if [ "${ENABLE_MORI}" != "1" ]; then \
|
||||||
|
echo "[MORI] Skipping (ENABLE_MORI=${ENABLE_MORI})"; \
|
||||||
|
exit 0; \
|
||||||
|
fi; \
|
||||||
|
echo "[MORI] Enabling MORI (NIC_BACKEND=${NIC_BACKEND})"; \
|
||||||
|
\
|
||||||
|
# Base deps for MORI build
|
||||||
|
apt-get update && apt-get install -y --no-install-recommends \
|
||||||
|
build-essential \
|
||||||
|
g++ \
|
||||||
|
jq \
|
||||||
|
libopenmpi-dev \
|
||||||
|
libpci-dev \
|
||||||
|
initramfs-tools \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*; \
|
||||||
|
\
|
||||||
|
# NIC backend deps
|
||||||
|
case "${NIC_BACKEND}" in \
|
||||||
|
# default: mlx5
|
||||||
|
none) \
|
||||||
|
export USE_IONIC="OFF"; \
|
||||||
|
export USE_BNXT="OFF"; \
|
||||||
|
;; \
|
||||||
|
# AMD NIC
|
||||||
|
ainic) \
|
||||||
|
export USE_IONIC="ON"; \
|
||||||
|
export USE_BNXT="OFF"; \
|
||||||
|
apt-get update && apt-get install -y --no-install-recommends ca-certificates curl gnupg apt-transport-https && \
|
||||||
|
rm -rf /var/lib/apt/lists/* && mkdir -p /etc/apt/keyrings; \
|
||||||
|
curl -fsSL https://repo.radeon.com/rocm/rocm.gpg.key | gpg --dearmor > /etc/apt/keyrings/amdainic.gpg; \
|
||||||
|
echo "deb [arch=amd64 signed-by=/etc/apt/keyrings/amdainic.gpg] https://repo.radeon.com/amdainic/pensando/ubuntu/${AINIC_VERSION} ${UBUNTU_CODENAME} main" \
|
||||||
|
> /etc/apt/sources.list.d/amdainic.list; \
|
||||||
|
apt-get update && apt-get install -y --no-install-recommends \
|
||||||
|
libionic-dev \
|
||||||
|
ionic-common \
|
||||||
|
; \
|
||||||
|
rm -rf /var/lib/apt/lists/*; \
|
||||||
|
;; \
|
||||||
|
# TODO: Add Broadcom bnxt packages/repos here later.
|
||||||
|
# bnxt) \
|
||||||
|
# export USE_IONIC="OFF"; \
|
||||||
|
# export USE_BNXT="ON"; \
|
||||||
|
# echo "[MORI] NIC_BACKEND=bnxt: USE_BNXT=ON. Add Broadcom bnxt packages/repos here later."; \
|
||||||
|
# ;; \
|
||||||
|
*) \
|
||||||
|
echo "ERROR: unknown NIC_BACKEND=${NIC_BACKEND}. Use one of: none, ainic"; \
|
||||||
|
exit 2; \
|
||||||
|
;; \
|
||||||
|
esac; \
|
||||||
|
\
|
||||||
|
# Build/install MORI
|
||||||
|
export MORI_GPU_ARCHS="${GPU_ARCH_LIST}"; \
|
||||||
|
echo "[MORI] MORI_GPU_ARCHS=${MORI_GPU_ARCHS} USE_IONIC=${USE_IONIC} USE_BNXT=${USE_BNXT}"; \
|
||||||
|
rm -rf /sgl-workspace/mori; \
|
||||||
|
git clone "${MORI_REPO}" /sgl-workspace/mori; \
|
||||||
|
cd /sgl-workspace/mori; \
|
||||||
|
git checkout "${MORI_COMMIT}"; \
|
||||||
|
git submodule update --init --recursive; \
|
||||||
|
python3 setup.py develop; \
|
||||||
|
python3 -c "import os, torch; print(os.path.join(os.path.dirname(torch.__file__), \"lib\"))" > /etc/ld.so.conf.d/torch.conf; \
|
||||||
|
ldconfig; \
|
||||||
|
echo "export PYTHONPATH=/sgl-workspace/mori:\${PYTHONPATH}" >> /etc/bash.bashrc; \
|
||||||
|
echo "[MORI] Done."'
|
||||||
|
|
||||||
# -----------------------
|
# -----------------------
|
||||||
# Performance environment variable.
|
# Performance environment variable.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,6 @@
|
|||||||
|
from sglang.srt.disaggregation.mori.conn import (
|
||||||
|
MoriKVBootstrapServer,
|
||||||
|
MoriKVManager,
|
||||||
|
MoriKVReceiver,
|
||||||
|
MoriKVSender,
|
||||||
|
)
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -231,6 +231,7 @@ class MetadataBuffers:
|
|||||||
|
|
||||||
class TransferBackend(Enum):
|
class TransferBackend(Enum):
|
||||||
MOONCAKE = "mooncake"
|
MOONCAKE = "mooncake"
|
||||||
|
MORI = "mori"
|
||||||
NIXL = "nixl"
|
NIXL = "nixl"
|
||||||
ASCEND = "ascend"
|
ASCEND = "ascend"
|
||||||
FAKE = "fake"
|
FAKE = "fake"
|
||||||
@@ -266,6 +267,23 @@ def get_kv_class(
|
|||||||
KVClassType.BOOTSTRAP_SERVER: MooncakeKVBootstrapServer,
|
KVClassType.BOOTSTRAP_SERVER: MooncakeKVBootstrapServer,
|
||||||
}
|
}
|
||||||
return class_mapping.get(class_type)
|
return class_mapping.get(class_type)
|
||||||
|
elif transfer_backend == TransferBackend.MORI:
|
||||||
|
from sglang.srt.disaggregation.base import KVArgs
|
||||||
|
from sglang.srt.disaggregation.mori import (
|
||||||
|
MoriKVBootstrapServer,
|
||||||
|
MoriKVManager,
|
||||||
|
MoriKVReceiver,
|
||||||
|
MoriKVSender,
|
||||||
|
)
|
||||||
|
|
||||||
|
class_mapping = {
|
||||||
|
KVClassType.KVARGS: KVArgs,
|
||||||
|
KVClassType.MANAGER: MoriKVManager,
|
||||||
|
KVClassType.SENDER: MoriKVSender,
|
||||||
|
KVClassType.RECEIVER: (MoriKVReceiver),
|
||||||
|
KVClassType.BOOTSTRAP_SERVER: MoriKVBootstrapServer,
|
||||||
|
}
|
||||||
|
return class_mapping.get(class_type)
|
||||||
elif transfer_backend == TransferBackend.ASCEND:
|
elif transfer_backend == TransferBackend.ASCEND:
|
||||||
from sglang.srt.disaggregation.ascend import (
|
from sglang.srt.disaggregation.ascend import (
|
||||||
AscendKVBootstrapServer,
|
AscendKVBootstrapServer,
|
||||||
|
|||||||
@@ -143,7 +143,7 @@ ATTENTION_BACKEND_CHOICES = [
|
|||||||
|
|
||||||
LORA_BACKEND_CHOICES = ["triton", "csgmv", "ascend", "torch_native"]
|
LORA_BACKEND_CHOICES = ["triton", "csgmv", "ascend", "torch_native"]
|
||||||
|
|
||||||
DISAGG_TRANSFER_BACKEND_CHOICES = ["mooncake", "nixl", "ascend", "fake"]
|
DISAGG_TRANSFER_BACKEND_CHOICES = ["mooncake", "nixl", "ascend", "fake", "mori"]
|
||||||
|
|
||||||
ENCODER_TRANSFER_BACKEND_CHOICES = ["zmq_to_scheduler", "zmq_to_tokenizer", "mooncake"]
|
ENCODER_TRANSFER_BACKEND_CHOICES = ["zmq_to_scheduler", "zmq_to_tokenizer", "mooncake"]
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,281 @@
|
|||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
import requests
|
||||||
|
|
||||||
|
from sglang.test.server_fixtures.disaggregation_fixture import (
|
||||||
|
PDDisaggregationServerBase,
|
||||||
|
)
|
||||||
|
from sglang.test.test_utils import (
|
||||||
|
DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
|
||||||
|
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
|
popen_launch_pd_server,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestMoriTransferEngineE2E(PDDisaggregationServerBase):
|
||||||
|
"""
|
||||||
|
Run:
|
||||||
|
SGLANG_MORI_MANUAL_E2E=1 python3 test/manual/test_mori_transfer_engine_e2e.py
|
||||||
|
|
||||||
|
Optional:
|
||||||
|
- SGLANG_MORI_E2E_TEST_MODEL: override model (defaults to a small test model)
|
||||||
|
- SGLANG_TEST_PD_DISAGG_DEVICES: RDMA devices string, e.g. "mlx5_roce0,mlx5_roce4"
|
||||||
|
"""
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def setUpClass(cls):
|
||||||
|
if os.environ.get("SGLANG_MORI_MANUAL_E2E", "") not in ("1", "true", "True"):
|
||||||
|
raise unittest.SkipTest(
|
||||||
|
"Set SGLANG_MORI_MANUAL_E2E=1 to run this manual MORI E2E test."
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
import torch
|
||||||
|
|
||||||
|
if not torch.cuda.is_available():
|
||||||
|
raise unittest.SkipTest("torch.cuda is not available.")
|
||||||
|
except Exception as e:
|
||||||
|
raise unittest.SkipTest(f"torch is not available/usable: {e}")
|
||||||
|
|
||||||
|
# Force the disaggregation fixture to use MORI backend in local/manual runs.
|
||||||
|
os.environ["SGLANG_TEST_PD_DISAGG_BACKEND"] = "mori"
|
||||||
|
|
||||||
|
super().setUpClass()
|
||||||
|
|
||||||
|
cls.model = os.environ.get(
|
||||||
|
"SGLANG_MORI_E2E_TEST_MODEL", DEFAULT_SMALL_MODEL_NAME_FOR_TEST
|
||||||
|
)
|
||||||
|
|
||||||
|
cls.start_prefill()
|
||||||
|
cls.start_decode()
|
||||||
|
|
||||||
|
cls.wait_server_ready(
|
||||||
|
cls.prefill_url + "/health", timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
|
||||||
|
)
|
||||||
|
cls.wait_server_ready(
|
||||||
|
cls.decode_url + "/health", timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
|
||||||
|
)
|
||||||
|
|
||||||
|
cls.launch_lb()
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def tearDownClass(cls):
|
||||||
|
os.environ.pop("SGLANG_TEST_PD_DISAGG_BACKEND", None)
|
||||||
|
super().tearDownClass()
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def launch_lb(cls):
|
||||||
|
lb_command = [
|
||||||
|
"python3",
|
||||||
|
"-m",
|
||||||
|
"sglang_router.launch_router",
|
||||||
|
"--pd-disaggregation",
|
||||||
|
"--mini-lb",
|
||||||
|
"--prefill",
|
||||||
|
cls.prefill_url,
|
||||||
|
"--decode",
|
||||||
|
cls.decode_url,
|
||||||
|
"--host",
|
||||||
|
cls.base_host,
|
||||||
|
"--port",
|
||||||
|
cls.lb_port,
|
||||||
|
]
|
||||||
|
print("Starting load balancer:", " ".join(lb_command))
|
||||||
|
cls.process_lb = subprocess.Popen(lb_command, stdout=None, stderr=None)
|
||||||
|
cls.wait_server_ready(
|
||||||
|
cls.lb_url + "/health", timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
|
||||||
|
)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def start_prefill(cls):
|
||||||
|
prefill_args = [
|
||||||
|
"--trust-remote-code",
|
||||||
|
"--disaggregation-mode",
|
||||||
|
"prefill",
|
||||||
|
"--tp",
|
||||||
|
"1",
|
||||||
|
]
|
||||||
|
prefill_args += cls.transfer_backend + cls.rdma_devices
|
||||||
|
cls.process_prefill = popen_launch_pd_server(
|
||||||
|
cls.model,
|
||||||
|
cls.prefill_url,
|
||||||
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
|
other_args=prefill_args,
|
||||||
|
)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def start_decode(cls):
|
||||||
|
decode_args = [
|
||||||
|
"--trust-remote-code",
|
||||||
|
"--disaggregation-mode",
|
||||||
|
"decode",
|
||||||
|
"--tp",
|
||||||
|
"1",
|
||||||
|
"--base-gpu-id",
|
||||||
|
"1",
|
||||||
|
]
|
||||||
|
decode_args += cls.transfer_backend + cls.rdma_devices
|
||||||
|
cls.process_decode = popen_launch_pd_server(
|
||||||
|
cls.model,
|
||||||
|
cls.decode_url,
|
||||||
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
|
other_args=decode_args,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_generate_smoke(self):
|
||||||
|
resp = requests.post(
|
||||||
|
self.lb_url + "/generate",
|
||||||
|
json={
|
||||||
|
"text": "Hello",
|
||||||
|
"sampling_params": {"temperature": 0, "max_new_tokens": 8},
|
||||||
|
},
|
||||||
|
timeout=120,
|
||||||
|
)
|
||||||
|
self.assertEqual(resp.status_code, 200, resp.text)
|
||||||
|
out = resp.json()
|
||||||
|
self.assertIn("text", out)
|
||||||
|
self.assertIsInstance(out["text"], str)
|
||||||
|
self.assertGreater(len(out["text"]), 0)
|
||||||
|
|
||||||
|
|
||||||
|
class TestMoriTransferEngineTPMismatchE2E(PDDisaggregationServerBase):
|
||||||
|
"""Manual MORI PD-disaggregation E2E with TP mismatch.
|
||||||
|
|
||||||
|
Scenario:
|
||||||
|
- prefill: tp=2 (GPU 0-1)
|
||||||
|
- decode: tp=4 (GPU 2-5)
|
||||||
|
|
||||||
|
Manual-only and requires >= 6 visible GPUs.
|
||||||
|
"""
|
||||||
|
|
||||||
|
_PORT_DELTA = 10
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def setUpClass(cls):
|
||||||
|
if os.environ.get("SGLANG_MORI_MANUAL_E2E", "") not in ("1", "true", "True"):
|
||||||
|
raise unittest.SkipTest(
|
||||||
|
"Set SGLANG_MORI_MANUAL_E2E=1 to run this manual MORI E2E test."
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
import torch
|
||||||
|
|
||||||
|
if not torch.cuda.is_available():
|
||||||
|
raise unittest.SkipTest("torch.cuda is not available.")
|
||||||
|
if torch.cuda.device_count() < 6:
|
||||||
|
raise unittest.SkipTest(
|
||||||
|
"TP-mismatch test requires >= 6 visible GPUs (prefill tp=2 + decode tp=4)."
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
raise unittest.SkipTest(f"torch is not available/usable: {e}")
|
||||||
|
|
||||||
|
os.environ["SGLANG_TEST_PD_DISAGG_BACKEND"] = "mori"
|
||||||
|
super().setUpClass()
|
||||||
|
|
||||||
|
# Shift ports to avoid clashing with TestMoriTransferEngineE2E.
|
||||||
|
cls.lb_port = str(int(cls.lb_port) + cls._PORT_DELTA)
|
||||||
|
cls.prefill_port = str(int(cls.prefill_port) + cls._PORT_DELTA)
|
||||||
|
cls.decode_port = str(int(cls.decode_port) + cls._PORT_DELTA)
|
||||||
|
cls.prefill_url = f"http://{cls.base_host}:{cls.prefill_port}"
|
||||||
|
cls.decode_url = f"http://{cls.base_host}:{cls.decode_port}"
|
||||||
|
cls.lb_url = f"http://{cls.base_host}:{cls.lb_port}"
|
||||||
|
|
||||||
|
cls.model = os.environ.get(
|
||||||
|
"SGLANG_MORI_E2E_TEST_MODEL", DEFAULT_SMALL_MODEL_NAME_FOR_TEST
|
||||||
|
)
|
||||||
|
|
||||||
|
cls.start_prefill()
|
||||||
|
cls.start_decode()
|
||||||
|
|
||||||
|
cls.wait_server_ready(
|
||||||
|
cls.prefill_url + "/health", timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
|
||||||
|
)
|
||||||
|
cls.wait_server_ready(
|
||||||
|
cls.decode_url + "/health", timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
|
||||||
|
)
|
||||||
|
cls.launch_lb()
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def tearDownClass(cls):
|
||||||
|
os.environ.pop("SGLANG_TEST_PD_DISAGG_BACKEND", None)
|
||||||
|
super().tearDownClass()
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def launch_lb(cls):
|
||||||
|
lb_command = [
|
||||||
|
"python3",
|
||||||
|
"-m",
|
||||||
|
"sglang_router.launch_router",
|
||||||
|
"--pd-disaggregation",
|
||||||
|
"--mini-lb",
|
||||||
|
"--prefill",
|
||||||
|
cls.prefill_url,
|
||||||
|
"--decode",
|
||||||
|
cls.decode_url,
|
||||||
|
"--host",
|
||||||
|
cls.base_host,
|
||||||
|
"--port",
|
||||||
|
cls.lb_port,
|
||||||
|
]
|
||||||
|
print("Starting load balancer:", " ".join(lb_command))
|
||||||
|
cls.process_lb = subprocess.Popen(lb_command, stdout=None, stderr=None)
|
||||||
|
cls.wait_server_ready(
|
||||||
|
cls.lb_url + "/health", timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
|
||||||
|
)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def start_prefill(cls):
|
||||||
|
prefill_args = [
|
||||||
|
"--trust-remote-code",
|
||||||
|
"--disaggregation-mode",
|
||||||
|
"prefill",
|
||||||
|
"--tp",
|
||||||
|
"2",
|
||||||
|
]
|
||||||
|
prefill_args += cls.transfer_backend + cls.rdma_devices
|
||||||
|
cls.process_prefill = popen_launch_pd_server(
|
||||||
|
cls.model,
|
||||||
|
cls.prefill_url,
|
||||||
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
|
other_args=prefill_args,
|
||||||
|
)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def start_decode(cls):
|
||||||
|
decode_args = [
|
||||||
|
"--trust-remote-code",
|
||||||
|
"--disaggregation-mode",
|
||||||
|
"decode",
|
||||||
|
"--tp",
|
||||||
|
"4",
|
||||||
|
"--base-gpu-id",
|
||||||
|
"2",
|
||||||
|
]
|
||||||
|
decode_args += cls.transfer_backend + cls.rdma_devices
|
||||||
|
cls.process_decode = popen_launch_pd_server(
|
||||||
|
cls.model,
|
||||||
|
cls.decode_url,
|
||||||
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
|
other_args=decode_args,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_generate_smoke_tp_mismatch(self):
|
||||||
|
resp = requests.post(
|
||||||
|
self.lb_url + "/generate",
|
||||||
|
json={
|
||||||
|
"text": "Hello",
|
||||||
|
"sampling_params": {"temperature": 0, "max_new_tokens": 8},
|
||||||
|
},
|
||||||
|
timeout=120,
|
||||||
|
)
|
||||||
|
self.assertEqual(resp.status_code, 200, resp.text)
|
||||||
|
out = resp.json()
|
||||||
|
self.assertIn("text", out)
|
||||||
|
self.assertIsInstance(out["text"], str)
|
||||||
|
self.assertGreater(len(out["text"]), 0)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
Reference in New Issue
Block a user