Fix Nightly NV CI (#33564)

Co-authored-by: Brayden Zhong <brayden@radixark.ai>
Co-authored-by: Baizhou Zhang <sobereddiezhang@gmail.com>
This commit is contained in:
Brayden Zhong
2026-08-05 18:06:47 -07:00
committed by GitHub
co-authored by Brayden Zhong Baizhou Zhang
parent e675c7226a
commit 28848bfe7c
4 changed files with 34 additions and 23 deletions
@@ -1725,6 +1725,9 @@ def _deepseek_moe_quant_resolution(view: Any) -> dict:
if ( if (
view.moe_a2a_backend == "none" view.moe_a2a_backend == "none"
and view.moe_runner_backend == "auto" and view.moe_runner_backend == "auto"
# LongCat top-k spans the zero-expert logits, which trtllm-gen's
# fused routing cannot see.
and not model_arch.startswith("LongcatFlash")
and ( and (
quantization quantization
in ["fp8", "modelopt_fp8", "modelopt_fp4", "modelopt_mixed"] in ["fp8", "modelopt_fp8", "modelopt_fp4", "modelopt_mixed"]
+1 -1
View File
@@ -49,7 +49,7 @@ class NgramEmbedding(torch.nn.Module):
+ int(over_embedding_m + i * 2 + 1) + int(over_embedding_m + i * 2 + 1)
) )
self.oe_embeder = VocabParallelEmbedding( self.oe_embeder = VocabParallelEmbedding(
num_embeddings=self.exclusive_oe_embedder_size_sums[-1], num_embeddings=int(self.exclusive_oe_embedder_size_sums[-1]),
embedding_dim=oe_hidden_dim, embedding_dim=oe_hidden_dim,
use_attn_tp_group=use_attn_tp_group, use_attn_tp_group=use_attn_tp_group,
) )
@@ -15,7 +15,7 @@ COMMON_ARGS = [
"--trust-remote-code", "--trust-remote-code",
"--reasoning-parser=glm45", "--reasoning-parser=glm45",
"--tool-call-parser=glm47", "--tool-call-parser=glm47",
"--mem-fraction-static=0.85", "--mem-fraction-static=0.8",
"--enable-metrics", "--enable-metrics",
] ]
@@ -6,8 +6,7 @@ from sglang.test.performance_test_runner import PerformanceTestParams
from sglang.test.run_combined_tests import run_combined_tests from sglang.test.run_combined_tests import run_combined_tests
from sglang.test.test_utils import ModelLaunchSettings from sglang.test.test_utils import ModelLaunchSettings
# Runs on both H200 and B200 via the nightly-8-gpu-common suite. register_cuda_ci(est_time=1200, suite="nightly-8-gpu-h200", nightly=True)
register_cuda_ci(est_time=1200, suite="nightly-8-gpu-common", nightly=True)
# LongCat-Flash-Lite-FP8 is the smallest member of the LongCat family # LongCat-Flash-Lite-FP8 is the smallest member of the LongCat family
# (~138 GB FP8 weights, hidden=3072, 14 layers, 256 routed + 128 zero # (~138 GB FP8 weights, hidden=3072, 14 layers, 256 routed + 128 zero
@@ -20,7 +19,6 @@ register_cuda_ci(est_time=1200, suite="nightly-8-gpu-common", nightly=True)
# LongcatFlashNgramForCausalLM is remapped to model_type "longcat_flash") # LongcatFlashNgramForCausalLM is remapped to model_type "longcat_flash")
# with LongCat-Flash-Chat-FP8 and LongCat-2.0-FP8, so it guards the shared # with LongCat-Flash-Chat-FP8 and LongCat-2.0-FP8, so it guards the shared
# code paths that recent LongCat EP fixes touched: # code paths that recent LongCat EP fixes touched:
# - the scheduler moe-topk gate for --moe-a2a-backend (PR #30975)
# - ScMoE dense-branch gather (RoPE) + the MoE-vs-DeepEPMoE double # - ScMoE dense-branch gather (RoPE) + the MoE-vs-DeepEPMoE double
# all_reduce fix (PR #31311) # all_reduce fix (PR #31311)
# - the zero-expert (identity) compute path (zero_expert_num=128) and # - the zero-expert (identity) compute path (zero_expert_num=128) and
@@ -41,28 +39,14 @@ class TestLongCatFlashLiteFp8(unittest.TestCase):
Two variants exercise the two MoE all-to-all backends that the LongCat Two variants exercise the two MoE all-to-all backends that the LongCat
EP fixes gate on: EP fixes gate on:
- EP8 + deepep : real expert parallelism (the path #30975/#31311 fix)
- EP8 + none : EP-over-TP baseline (all_reduce / gather correctness) - EP8 + none : EP-over-TP baseline (all_reduce / gather correctness)
- EP8 + deepep : real expert parallelism (the path #30975/#31311 fix),
skipped until DeepEP raises its low-latency top-k cap to 12
""" """
def test_longcat_flash_lite_fp8(self): def _run(self, variant: ModelLaunchSettings):
variants = [
ModelLaunchSettings(
LONGCAT_FLASH_LITE_FP8_MODEL_PATH,
tp_size=8,
extra_args=COMMON_ARGS + ["--ep=8"],
variant="TP8+EP8+none",
),
ModelLaunchSettings(
LONGCAT_FLASH_LITE_FP8_MODEL_PATH,
tp_size=8,
extra_args=COMMON_ARGS + ["--ep=8", "--moe-a2a-backend=deepep"],
variant="TP8+EP8+deepep",
),
]
run_combined_tests( run_combined_tests(
models=variants, models=[variant],
test_name="LongCat-Flash-Lite-FP8", test_name="LongCat-Flash-Lite-FP8",
# Measured 2026-07-22 on 8xH100-80GB, gsm8k 200q, 5-shot, greedy: # Measured 2026-07-22 on 8xH100-80GB, gsm8k 200q, 5-shot, greedy:
# TP8+EP8+none -> 0.840 # TP8+EP8+none -> 0.840
@@ -78,6 +62,30 @@ class TestLongCatFlashLiteFp8(unittest.TestCase):
), ),
) )
def test_longcat_flash_lite_fp8(self):
self._run(
ModelLaunchSettings(
LONGCAT_FLASH_LITE_FP8_MODEL_PATH,
tp_size=8,
extra_args=COMMON_ARGS + ["--ep=8"],
variant="TP8+EP8+none",
)
)
@unittest.skip(
"Blocked: DeepEP low-latency dispatch asserts num_topk <= kNumMaxTopK "
"(11 in internode_ll.cu), LongCat moe_topk is 12."
)
def test_longcat_flash_lite_fp8_deepep(self):
self._run(
ModelLaunchSettings(
LONGCAT_FLASH_LITE_FP8_MODEL_PATH,
tp_size=8,
extra_args=COMMON_ARGS + ["--ep=8", "--moe-a2a-backend=deepep"],
variant="TP8+EP8+deepep",
)
)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()