diff --git a/docs_new/cookbook/autoregressive/Qwen/Qwen3.5.mdx b/docs_new/cookbook/autoregressive/Qwen/Qwen3.5.mdx index 895360dc5..023360f42 100644 --- a/docs_new/cookbook/autoregressive/Qwen/Qwen3.5.mdx +++ b/docs_new/cookbook/autoregressive/Qwen/Qwen3.5.mdx @@ -127,7 +127,7 @@ This section provides deployment configurations optimized for different hardware - Speculative decoding (MTP) can significantly reduce latency for interactive use cases. - **H100 FP8:** Add `--enable-symm-mem` to enable NCCL symmetric memory for faster collectives and better performance under multi-GPU settings. -- **AMD GPUs (MI300X / MI325X / MI355X):** Use `SGLANG_USE_AITER=1` and `SGLANG_USE_AITER_UNIFIED_ATTN=1` with `--attention-backend aiter`, which requires `--page-size 16` and can also enable `--enable-aiter-allreduce-fusion`. Additionally set `AITER_FLYDSL_FORCE=1` to force the AITER FlyDSL MoE kernels and `SGLANG_MAMBA_SSM_DTYPE=bfloat16` to store the Mamba SSM state in bfloat16 (instead of the default float32). +- **AMD GPUs (MI300X / MI325X / MI355X):** Use `SGLANG_USE_AITER=1` and `SGLANG_USE_AITER_UNIFIED_ATTN=1` with `--attention-backend aiter`, which requires `--page-size 16` and can also enable `--enable-aiter-allreduce-fusion`. Additionally set `AITER_FLYDSL_FORCE=1` to force the AITER FlyDSL MoE kernels and `SGLANG_MAMBA_SSM_DTYPE=bfloat16` to store the Mamba SSM state in bfloat16 (instead of the default float32). For the **MXFP4 checkpoint on MI355X**, set `ROCM_QUICK_REDUCE_QUANTIZATION=INT8` to route multi-GPU collectives through INT8-quantized ROCm quick all-reduce, and drop `--enable-aiter-allreduce-fusion` (the two are mutually exclusive; quick all-reduce is preferred for this recipe). - **Watchdog timeout:** Increase `--watchdog-timeout` to `1200` or higher for this large model, as weight loading can take significant time. - **Mamba Radix Cache**: Qwen3.5's hybrid Gated Delta Networks architecture supports two mamba scheduling strategies via `--mamba-radix-cache-strategy`: - **V1 (`no_buffer`)**: Default. No overlap scheduler, lower memory usage. Required for AMD MI GPUs. diff --git a/docs_new/src/snippets/autoregressive/qwen35-deployment.jsx b/docs_new/src/snippets/autoregressive/qwen35-deployment.jsx index 69795a7a0..eebdc8881 100644 --- a/docs_new/src/snippets/autoregressive/qwen35-deployment.jsx +++ b/docs_new/src/snippets/autoregressive/qwen35-deployment.jsx @@ -433,16 +433,22 @@ export const Qwen35Deployment = () => { // Append AMD GPU-specific backend configurations. // All AMD MI GPUs use the AITER unified-attention backend (pair with // SGLANG_USE_AITER=1 and SGLANG_USE_AITER_UNIFIED_ATTN=1; see cookbook prose), - // which requires --page-size 16. Enable AITER allreduce fusion for multi-GPU. + // which requires --page-size 16. Multi-GPU runs enable AITER allreduce fusion, + // except the MXFP4 MI355X recipe, which uses ROCm INT8 quantized quick + // all-reduce (ROCM_QUICK_REDUCE_QUANTIZATION=INT8) instead. if (amdGpu) { + const amdFp4 = quantization === 'fp4' && hardware === 'mi355x'; let amdEnv = "SGLANG_USE_AITER=1 \\\nSGLANG_USE_AITER_UNIFIED_ATTN=1 \\\nAITER_FLYDSL_FORCE=1 \\\n"; if (MOE_MODELS.has(model)) { amdEnv += "SGLANG_MAMBA_SSM_DTYPE=bfloat16 \\\n"; } + if (amdFp4) { + amdEnv += "ROCM_QUICK_REDUCE_QUANTIZATION=INT8 \\\n"; + } cmd = amdEnv + cmd; cmd += " \\\n --attention-backend aiter"; cmd += " \\\n --page-size 16"; - if (hwConfig.tp > 1) { + if (hwConfig.tp > 1 && !amdFp4) { cmd += " \\\n --enable-aiter-allreduce-fusion"; } } @@ -464,8 +470,10 @@ export const Qwen35Deployment = () => { // FP4-specific backend settings if (quantization === 'fp4') { if (hardware === 'mi355x') { - // AMD MXFP4 on MI355X: backend / --page-size 16 / AITER allreduce fusion - // are emitted by the AMD backend block above. Add the FP4-specific flags here. + // AMD MXFP4 on MI355X: backend / --page-size 16 and the INT8 quantized + // ROCm quick all-reduce env are emitted by the AMD backend block above + // (this recipe uses quick all-reduce instead of AITER allreduce fusion). + // Add the FP4-specific flags here. cmd += ' \\\n --disable-radix-cache'; // Cap concurrency under MTP to avoid OOM at tp=2. if (speculative === 'enabled') {