[Cookbook][CPU]Update CPU model support info in Cookbook (#30520)

Co-authored-by: zijiexia <37504505+zijiexia@users.noreply.github.com>
This commit is contained in:
Zaili Wang
2026-07-16 03:42:09 +00:00
committed by GitHub
co-authored by zijiexia
parent dee91c51cf
commit bc525dcf90
23 changed files with 221 additions and 67 deletions
@@ -49,7 +49,7 @@ This section provides deployment configurations optimized for different hardware
- **Single GPU Deployment:** DeepSeek-OCR-2 (~3B parameters) fits on a single modern GPU — no tensor parallelism required for low-latency serving.
- **High Throughput:** For high-throughput scenarios, use data parallelism with the SGLang Model Gateway. See [DP, DPA and SGLang DP Router](../../../docs/advanced_features/sgl_model_gateway).
- **NCCL timeout:** If model loading is slow, increase `--dist-timeout 3600`.
- For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for NUMA binding settings.
- **Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for NUMA binding settings.
## 4. Model Invocation
@@ -99,7 +99,7 @@ python3 -m sglang.compile_deep_gemm --model deepseek-ai/DeepSeek-R1 --tp 8 --tru
**NCCL timeout:** If model loading is slow, increase: `--dist-timeout 3600`.
For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
**Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -79,7 +79,7 @@ DeepGEMM is enabled by default on Hopper/Blackwell and can be disabled with `SGL
**NCCL timeout:** If model loading is slow and you hit an NCCL timeout, increase it: `--dist-timeout 3600`.
For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
**Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -77,7 +77,7 @@ DeepGEMM is enabled by default on Hopper/Blackwell and can be disabled with `SGL
**NCCL timeout:** If model loading is slow and you hit an NCCL timeout, increase it: `--dist-timeout 3600`.
For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
**Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -22,6 +22,8 @@ SGLang offers multiple installation methods. You can choose the most suitable in
Please refer to the [official SGLang installation guide](../../../docs/get-started/install) for installation instructions.
For SGLang CPU installation, please refer to the [CPU version installation guide](../../../docs/hardware-platforms/cpu_server#installation).
## 3. Model Deployment
This section provides deployment configurations optimized for different hardware platforms and use cases.
@@ -55,6 +57,10 @@ import { Llama31Deployment } from "/src/snippets/autoregressive/llama31-deployme
- 70B/8B: Use AMD's optimized `amd/Llama-3.1-{size}-Instruct-FP8-KV`
- **Tool Calling**: Enable with `--tool-call-parser llama3` for Instruct models
**Xeon CPU Deployment:**
- Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
### 4.1 Basic Usage
@@ -24,9 +24,11 @@ For more details, please refer to the [official Llama models repository](https:/
Please refer to the [official SGLang installation guide](../../../docs/get-started/install) for installation instructions.
For SGLang CPU installation, please refer to the [CPU version installation guide](../../../docs/hardware-platforms/cpu_server#installation).
## 3. Model Deployment
This section provides deployment configurations optimized for AMD GPUs (MI300X, MI325X, MI355X).
This section provides deployment configurations optimized for AMD GPUs (MI300X, MI325X, MI355X) and Intel Xeon CPUs.
### 3.1 Interactive Configuration
@@ -45,6 +47,10 @@ import { Llama33Deployment } from "/src/snippets/autoregressive/llama33-70b-depl
- **Tool Calling**: Enable with `--tool-call-parser llama3` for function calling support
- **Higher Throughput**: Optional TP=2 or TP=4 can be used for increased throughput
**Xeon CPU Deployment:**
Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
### 4.1 Basic Usage
@@ -29,6 +29,8 @@ SGLang offers multiple installation methods. You can choose the most suitable in
Please refer to the [official SGLang installation guide](../../../docs/get-started/install) for installation instructions.
For SGLang CPU installation, please refer to the [CPU version installation guide](../../../docs/hardware-platforms/cpu_server#installation).
## 3. Model Deployment
This section provides a progressive guide from quick deployment to performance optimization, suitable for users at different levels.
@@ -54,6 +56,7 @@ This section provides a progressive guide from quick deployment to performance o
- **Multi-Modal:** Add `--enable-multimodal` to enable image input support.
- **Hybrid KV Cache:** Set `--swa-full-tokens-ratio` to control the ratio of SWA (local attention) KV tokens to full-attention KV tokens (default: 0.8, range: 0–1).
- **EAGLE Speculative Decoding:** Supported for Llama 4 Scout and Maverick via EAGLE3. Enable with the interactive command generator above.
- **Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -25,6 +25,8 @@ SGLang offers multiple installation methods. You can choose the most suitable in
Please refer to the [official SGLang installation guide](../../../docs/get-started/install) for installation instructions.
For SGLang CPU installation, please refer to the [CPU version installation guide](../../../docs/hardware-platforms/cpu_server#installation).
## 3.Model Deployment
This section provides deployment configurations optimized for different hardware platforms and use cases.
@@ -47,6 +49,7 @@ import { GPTOSSDeployment } from "/src/snippets/autoregressive/gpt-oss-deploymen
- **MCP tool servers:** For production, point SGLang at external MCP SSE servers with `--tool-server ip-1:port-1,ip-2:port-2`.
- **Responses API:** GPT-OSS supports OpenAI's Responses API (`client.responses.create`) in addition to the standard Chat Completions API (see section 4.2.4).
- **Use Python 3.12** when running the demo Python tool.
- **Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4.Model Invocation
@@ -29,13 +29,15 @@ SGLang offers multiple installation methods. You can choose the most suitable in
Please refer to the [official SGLang installation guide](../../../docs/get-started/install) for installation instructions.
For SGLang CPU installation, please refer to the [CPU version installation guide](../../../docs/hardware-platforms/cpu_server#installation).
## 3. Model Deployment
This section provides deployment configurations optimized for AMD MI300X, MI325X and MI355X hardware platforms and different use cases.
This section provides deployment configurations optimized for AMD MI300X, MI325X and MI355X as well as Intel Xeon CPU hardware platforms and different use cases.
### 3.1 Basic Configuration
The Qwen2.5-VL series offers models in various sizes. The following configurations have been verified on AMD MI300X, MI325X and MI355X GPUs.
The Qwen2.5-VL series offers models in various sizes. The following configurations have been verified on AMD MI300X, MI325X and MI355X GPUs as well as Intel Xeon CPUs.
**Interactive Command Generator**: Use the configuration selector below to automatically generate the appropriate deployment command for your hardware platform and model size.
@@ -45,6 +47,7 @@ The Qwen2.5-VL series offers models in various sizes. The following configuratio
* **Memory Management**: For the 72B model on MI300X/MI325X/MI355X, we have verified successful deployment with `--context-length 128000`. Smaller context lengths can be used to reduce memory usage if needed.
* **Multi-GPU Deployment**: Use Tensor Parallelism (`--tp`) to scale across multiple GPUs. For example, use `--tp 8` for the 72B model and `--tp 2` for the 32B model on MI300X/MI325X/MI355X.
* **Xeon CPU service configuration**: Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -48,7 +48,7 @@ This section provides a progressive guide from quick deployment to performance o
- **Mamba Radix Cache**: Qwen3-Coder-Next's hybrid Gated Delta Networks architecture supports two mamba scheduling strategies via `--mamba-scheduler-strategy`:
- **V1 (`no_buffer`)**: Default. No overlap scheduler, lower memory usage.
- **V2 (`extra_buffer`)**: Enables overlap scheduling and branching point caching with `--mamba-scheduler-strategy extra_buffer --page-size 64`. Requires FLA kernel backend. Trades higher mamba state memory for better throughput. Strictly superior in non-KV-cache-bound scenarios; in KV-cache-bound cases, weigh the overlap scheduling benefit against reduced max concurrency. `--page-size` must satisfy `FLA_CHUNK_SIZE % page_size == 0` or `page_size % FLA_CHUNK_SIZE == 0` (`FLA_CHUNK_SIZE` is currently 64).
- For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
- **Xeon CPU service configuration**: Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -53,12 +53,12 @@ This section provides deployment configurations verified on AMD MI300X, MI325X,
* **NVFP4 Quantization**: Requires `--quantization modelopt_fp4` and uses a different model path (`nvidia/Qwen3-Coder-...`).
* **DP Attention**: NVFP4 configuration supports `--enable-dp-attention` for improved throughput.
**Intel Xeon CPU:**
* Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
**General:**
* **Tool Use**: To enable tool calling capabilities, add `--tool-call-parser qwen3_coder` to the launch command.
**Intel Xeon CPU:**
* For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
### 4.1 Basic Usage
@@ -58,7 +58,7 @@ The Qwen3-Next series comes in only one size but offers different thinking modes
- **V1 (`no_buffer`)**: Default. No overlap scheduler, lower memory usage.
- **V2 (`extra_buffer`)**: Enables overlap scheduling and branching point caching with `--mamba-scheduler-strategy extra_buffer --page-size 64`. Requires FLA kernel backend. Trades higher mamba state memory for better throughput. Strictly superior in non-KV-cache-bound scenarios; in KV-cache-bound cases, weigh the overlap scheduling benefit against reduced max concurrency. `--page-size` must satisfy `FLA_CHUNK_SIZE % page_size == 0` or `page_size % FLA_CHUNK_SIZE == 0` (`FLA_CHUNK_SIZE` is currently 64).
- For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
- **Xeon CPU service configuration**: Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -78,7 +78,7 @@ python -m sglang.launch_server \
--enable-metrics
```
* For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
* **Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -228,7 +228,7 @@ This section provides deployment configurations optimized for different hardware
**FP8 KV Cache**: `--kv-cache-dtype fp8_e4m3` quantizes the KV cache to FP8 at runtime. Since these FP8 model checkpoints do not include pre-calibrated KV cache scaling factors, SGLang defaults to a scale of 1.0, which may cause noticeable accuracy degradation on reasoning-heavy tasks. It is not included in the generated commands above; add it manually only if memory constraints require the trade-off.
</Warning>
- For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
- **Xeon CPU service configuration**: Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -151,7 +151,7 @@ All Qwen3.6 variants (MoE 35B-A3B and Dense 27B) fit on a single supported GPU a
</table>
- For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
- **Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -44,16 +44,16 @@ import { Qwen3Deployment } from "/src/snippets/autoregressive/qwen3-deployment.j
### 3.2 Configuration Tips
- **Memory Management** : Set lower `--context-length` to conserve memory. A value of `128000` is sufficient for most scenarios, down from the default 262K.
- **Expert Parallelism** : SGLang supports Expert Parallelism (EP) via `--ep`, allowing experts in MoE models to be deployed on separate GPUs for better throughput. One thing to note is that, for quantized models, you need to set `--ep` to a value that satisfies the requirement: `(moe_intermediate_size / moe_tp_size) % weight_block_size_n == 0, where moe_tp_size is equal to tp_size divided by ep_size.` Note that EP may perform worse in low concurrency scenarios due to additional communication overhead. Check out [Expert Parallelism Deployment](../../../docs/advanced_features/expert_parallelism) for more details.
- **Kernel Tuning** : For MoE Triton kernel tuning on your specific hardware, refer to [fused_moe_triton](https://github.com/sgl-project/sglang/tree/main/benchmark/kernels/fused_moe_triton).
- **Speculative Decoding**: Using Speculative Decoding for latency-sensitive scenarios.
- **Memory Management:** Set lower `--context-length` to conserve memory. A value of `128000` is sufficient for most scenarios, down from the default 262K.
- **Expert Parallelism:** SGLang supports Expert Parallelism (EP) via `--ep`, allowing experts in MoE models to be deployed on separate GPUs for better throughput. One thing to note is that, for quantized models, you need to set `--ep` to a value that satisfies the requirement: `(moe_intermediate_size / moe_tp_size) % weight_block_size_n == 0, where moe_tp_size is equal to tp_size divided by ep_size.` Note that EP may perform worse in low concurrency scenarios due to additional communication overhead. Check out [Expert Parallelism Deployment](../../../docs/advanced_features/expert_parallelism) for more details.
- **Kernel Tuning:** For MoE Triton kernel tuning on your specific hardware, refer to [fused_moe_triton](https://github.com/sgl-project/sglang/tree/main/benchmark/kernels/fused_moe_triton).
- **Speculative Decoding:** Using Speculative Decoding for latency-sensitive scenarios.
- `--speculative-algorithm EAGLE3`: Speculative decoding algorithm
- `--speculative-num-steps 3`: Number of speculative verification rounds
- `--speculative-eagle-topk 1`: Top-k sampling for draft tokens
- `--speculative-num-draft-tokens 4`: Number of draft tokens per step
- `--speculative-draft-model-path`: The path of the draft model weights. This can be a local folder or a Hugging Face repo ID such as [`lmsys/SGLang-EAGLE3-Qwen3-235B-A22B-Instruct-2507-SpecForge-Meituan`](https://huggingface.co/lmsys/SGLang-EAGLE3-Qwen3-235B-A22B-Instruct-2507-SpecForge-Meituan).
- For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
- **Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -168,7 +168,7 @@ sglang serve \
Toggle the "Speculative Decoding (MTP)" option in the generator above to add these flags automatically. Tune `num-steps` / `num-draft-tokens` based on acceptance rate in your workload.
For configuring CPU service, please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
**Xeon CPU service configuration:** Please refer to the `Notes` part in the serving engine launching section in [the SGLang CPU server document](../../../docs/hardware-platforms/cpu_server#launch-of-the-serving-engine) to better understand how to configure the arguments, especially for TP (tensor parallel) and NUMA binding settings.
## 4. Model Invocation
@@ -11,7 +11,8 @@ export const GPTOSSDeployment = () => {
{ id: 'h100', label: 'H100', default: false },
{ id: 'mi300x', label: 'MI300X', default: false },
{ id: 'mi325x', label: 'MI325X', default: false },
{ id: 'mi355x', label: 'MI355X', default: false }
{ id: 'mi355x', label: 'MI355X', default: false },
{ id: 'xeon', label: 'XEON', default: false }
]
},
modelsize: {
@@ -56,6 +57,18 @@ export const GPTOSSDeployment = () => {
}
};
const getDisplayOptions = (values) => ({
...options,
quantization: options.quantization,
speculative: {
...options.speculative,
items: options.speculative.items.map(item => ({
...item,
disabled: values.hardware === 'xeon' && item.id === 'enabled'
}))
}
});
// Initialize state
const getInitialState = () => {
const initialState = {};
@@ -89,7 +102,13 @@ export const GPTOSSDeployment = () => {
}, []);
const handleRadioChange = (optionName, value) => {
setValues(prev => ({ ...prev, [optionName]: value }));
setValues(prev => {
const next = { ...prev, [optionName]: value };
if (optionName === 'hardware' && value === 'xeon') {
next.speculative = 'disabled';
}
return next;
});
};
// Generate command
@@ -106,7 +125,8 @@ export const GPTOSSDeployment = () => {
b300: { tp: 8 },
mi300x: { tp: 8 },
mi325x: { tp: 8 },
mi355x: { tp: 8 }
mi355x: { tp: 8 },
xeon: { tp: 3 }
},
'20b': {
baseName: '20b',
@@ -116,7 +136,8 @@ export const GPTOSSDeployment = () => {
b300: { tp: 1 },
mi300x: { tp: 1 },
mi325x: { tp: 1 },
mi355x: { tp: 1 }
mi355x: { tp: 1 },
xeon: { tp: 3 }
}
};
@@ -146,6 +167,7 @@ export const GPTOSSDeployment = () => {
return '# MI300X/MI325X GPUs with MXFP4 quantization: Work In Progress';
}
// AMD MI30x requires SGLANG_USE_AITER=0 due to YaRN RoPE precision issues
if (hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi355x') {
cmd += 'SGLANG_USE_AITER=0 ';
@@ -159,6 +181,12 @@ export const GPTOSSDeployment = () => {
cmd += ` --model ${modelName}`;
if (hardware === 'xeon') {
cmd += ` \\
--device cpu \\
--disable-overlap-schedule`;
}
if (hwConfig.tp > 1) {
cmd += ` \\\n --tp ${hwConfig.tp}`;
}
@@ -206,7 +234,7 @@ export const GPTOSSDeployment = () => {
return (
<div style={containerStyle} className="not-prose">
{Object.entries(options).map(([key, option]) => (
{Object.entries(getDisplayOptions(values)).map(([key, option]) => (
<div key={key} style={cardStyle}>
<div style={titleStyle}>{option.title}</div>
<div style={itemsStyle}>
@@ -225,9 +253,10 @@ export const GPTOSSDeployment = () => {
) : (
option.items.map(item => {
const isChecked = values[option.name] === item.id;
const isDisabled = Boolean(item.disabled);
return (
<label key={item.id} style={{ ...labelBaseStyle, ...(isChecked ? checkedStyle : {}) }}>
<input type="radio" name={option.name} value={item.id} checked={isChecked} onChange={() => handleRadioChange(option.name, item.id)} style={{ display: 'none' }} />
<label key={item.id} style={{ ...labelBaseStyle, ...(isChecked ? checkedStyle : {}), ...(isDisabled ? disabledStyle : {}) }}>
<input type="radio" name={option.name} value={item.id} checked={isChecked} disabled={isDisabled} onChange={() => !isDisabled && handleRadioChange(option.name, item.id)} style={{ display: 'none' }} />
{item.label}
{item.subtitle && <small style={{ ...subtitleStyle, color: isChecked ? 'rgba(255,255,255,0.85)' : 'inherit' }}>{item.subtitle}</small>}
</label>
@@ -10,7 +10,8 @@ export const Llama31Deployment = () => {
{ id: 'b200', label: 'B200', default: false },
{ id: 'mi300x', label: 'MI300X', default: false },
{ id: 'mi325x', label: 'MI325X', default: false },
{ id: 'mi355x', label: 'MI355X', default: false }
{ id: 'mi355x', label: 'MI355X', default: false },
{ id: 'xeon', label: 'XEON', default: false }
]
},
modelsize: {
@@ -57,6 +58,34 @@ export const Llama31Deployment = () => {
}
};
const getDisplayOptions = (values) => {
const displayOptions = {
...options,
modelsize: {
...options.modelsize,
items: options.modelsize.items.map(item => ({
...item,
disabled: values.hardware === 'xeon' && item.id !== '8b'
}))
},
quantization: {
...options.quantization,
items: options.quantization.items.map(item => ({
...item,
disabled: values.hardware === 'xeon' && item.id === 'fp8'
}))
},
optimization: {
...options.optimization,
items: options.optimization.items.map(item => ({
...item,
disabled: values.hardware === 'xeon' && item.id !== 'basic'
}))
}
};
return displayOptions;
};
// Initialize state
const getInitialState = () => {
const initialState = {};
@@ -86,7 +115,15 @@ export const Llama31Deployment = () => {
}, []);
const handleRadioChange = (optionName, value) => {
setValues(prev => ({ ...prev, [optionName]: value }));
setValues(prev => {
const next = { ...prev, [optionName]: value };
if (optionName === 'hardware' && value === 'xeon') {
next.modelsize = '8b';
next.quantization = 'bf16';
next.optimization = 'basic';
}
return next;
});
};
// Generate command
@@ -94,6 +131,8 @@ export const Llama31Deployment = () => {
const { hardware, optimization, modelsize, category, toolcall, quantization } = values;
const isAMD = hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi355x';
const isXeon = hardware === 'xeon';
const effectiveModelSize = isXeon ? '8b' : modelsize;
// Model size mapping
const sizeMap = {
@@ -101,13 +140,13 @@ export const Llama31Deployment = () => {
'70b': '70B',
'405b': '405B'
};
const sizeToken = sizeMap[modelsize] || '70B';
const sizeToken = sizeMap[effectiveModelSize] || '70B';
const categorySuffix = category === 'instruct' ? '-Instruct' : '';
// Determine model path
let modelPath;
if (quantization === 'fp8' && category === 'instruct') {
if (modelsize === '405b') {
if (quantization === 'fp8' && category === 'instruct' && !isXeon) {
if (effectiveModelSize === '405b') {
// Meta official FP8 for 405B
modelPath = `meta-llama/Llama-3.1-${sizeToken}${categorySuffix}-FP8`;
} else if (isAMD) {
@@ -142,13 +181,16 @@ export const Llama31Deployment = () => {
}
};
tpSize = quantization === 'fp8'
? amdTpConfig[hardware][modelsize].fp8
: amdTpConfig[hardware][modelsize].bf16;
? amdTpConfig[hardware][effectiveModelSize].fp8
: amdTpConfig[hardware][effectiveModelSize].bf16;
} else if (isXeon) {
// Intel Xeon CPU TP configuration
tpSize = 3;
} else {
// NVIDIA GPU TP configuration
if (modelsize === '405b') {
if (effectiveModelSize === '405b') {
tpSize = 8;
} else if (modelsize === '70b' && (hardware === 'h100' || hardware === 'h200')) {
} else if (effectiveModelSize === '70b' && (hardware === 'h100' || hardware === 'h200')) {
tpSize = 2;
}
}
@@ -157,17 +199,22 @@ export const Llama31Deployment = () => {
const args = [];
args.push(`--model-path ${modelPath}`);
if (isXeon) {
args.push(`--device cpu`);
args.push(`--disable-overlap-schedule`);
}
if (tpSize) {
args.push(`--tp ${tpSize}`);
}
// Add quantization flag only if not using FP8 variant model
if (quantization === 'fp8' && category !== 'instruct') {
if (quantization === 'fp8' && category !== 'instruct' && !isXeon) {
args.push(`--quantization fp8`);
}
// NVIDIA-specific optimizations
if (!isAMD) {
if (!isAMD && !isXeon) {
if (optimization === 'throughput') {
args.push(`--enable-dp-attention`);
args.push(`--mem-fraction-static 0.85`);
@@ -176,7 +223,7 @@ export const Llama31Deployment = () => {
args.push(`--speculative-num-steps 3`);
args.push(`--speculative-eagle-topk 1`);
args.push(`--speculative-num-draft-tokens 4`);
if (modelsize === '8b' && category === 'instruct') {
if (effectiveModelSize === '8b' && category === 'instruct') {
args.push(`--speculative-draft-model-path yuhuili/EAGLE3-LLaMA3.1-Instruct-8B`);
} else {
args.push(`--speculative-draft-model-path \${EAGLE3_MODEL_PATH}`);
@@ -212,7 +259,7 @@ export const Llama31Deployment = () => {
return (
<div style={containerStyle} className="not-prose">
{Object.entries(options).map(([key, option]) => (
{Object.entries(getDisplayOptions(values)).map(([key, option]) => (
<div key={key} style={cardStyle}>
<div style={titleStyle}>{option.title}</div>
<div style={itemsStyle}>
@@ -231,9 +278,10 @@ export const Llama31Deployment = () => {
) : (
option.items.map(item => {
const isChecked = values[option.name] === item.id;
const isDisabled = Boolean(item.disabled);
return (
<label key={item.id} style={{ ...labelBaseStyle, ...(isChecked ? checkedStyle : {}) }}>
<input type="radio" name={option.name} value={item.id} checked={isChecked} onChange={() => handleRadioChange(option.name, item.id)} style={{ display: 'none' }} />
<label key={item.id} style={{ ...labelBaseStyle, ...(isChecked ? checkedStyle : {}), ...(isDisabled ? disabledStyle : {}) }}>
<input type="radio" name={option.name} value={item.id} checked={isChecked} disabled={isDisabled} onChange={() => !isDisabled && handleRadioChange(option.name, item.id)} style={{ display: 'none' }} />
{item.label}
{item.subtitle && <small style={{ ...subtitleStyle, color: isChecked ? 'rgba(255,255,255,0.85)' : 'inherit' }}>{item.subtitle}</small>}
</label>
@@ -7,7 +7,8 @@ export const Llama33Deployment = () => {
items: [
{ id: 'mi300x', label: 'MI300X', default: true },
{ id: 'mi325x', label: 'MI325X', default: false },
{ id: 'mi355x', label: 'MI355X', default: false }
{ id: 'mi355x', label: 'MI355X', default: false },
{ id: 'xeon', label: 'XEON', default: false }
]
},
quantization: {
@@ -28,6 +29,17 @@ export const Llama33Deployment = () => {
}
};
const getDisplayOptions = (values) => ({
...options,
quantization: {
...options.quantization,
items: options.quantization.items.map(item => ({
...item,
disabled: values.hardware === 'xeon' && item.id === 'fp8'
}))
}
});
// Initialize state
const getInitialState = () => {
const initialState = {};
@@ -57,7 +69,13 @@ export const Llama33Deployment = () => {
}, []);
const handleRadioChange = (optionName, value) => {
setValues(prev => ({ ...prev, [optionName]: value }));
setValues(prev => {
const next = { ...prev, [optionName]: value };
if (optionName === 'hardware' && value === 'xeon') {
next.quantization = 'bf16';
}
return next;
});
};
// Generate command
@@ -65,14 +83,20 @@ export const Llama33Deployment = () => {
const { hardware, quantization, toolcall } = values;
// Select model based on quantization
const modelPath = quantization === 'fp8'
const modelPath = quantization === 'fp8' && hardware !== 'xeon'
? 'amd/Llama-3.3-70B-Instruct-FP8-KV'
: 'meta-llama/Llama-3.3-70B-Instruct';
// Build command
let cmd = 'python -m sglang.launch_server \\\n';
cmd += ` --model-path ${modelPath} \\\n`;
cmd += ` --tp 1`;
if (hardware === 'xeon') {
cmd += ` --device cpu \\\n`;
cmd += ` --disable-overlap-schedule \\\n`;
cmd += ` --tp 6`;
} else {
cmd += ` --tp 1`;
}
// Add tool calling parser
if (toolcall === 'enabled') {
@@ -98,7 +122,7 @@ export const Llama33Deployment = () => {
return (
<div style={containerStyle} className="not-prose">
{Object.entries(options).map(([key, option]) => (
{Object.entries(getDisplayOptions(values)).map(([key, option]) => (
<div key={key} style={cardStyle}>
<div style={titleStyle}>{option.title}</div>
<div style={itemsStyle}>
@@ -117,9 +141,10 @@ export const Llama33Deployment = () => {
) : (
option.items.map(item => {
const isChecked = values[option.name] === item.id;
const isDisabled = Boolean(item.disabled);
return (
<label key={item.id} style={{ ...labelBaseStyle, ...(isChecked ? checkedStyle : {}) }}>
<input type="radio" name={option.name} value={item.id} checked={isChecked} onChange={() => handleRadioChange(option.name, item.id)} style={{ display: 'none' }} />
<label key={item.id} style={{ ...labelBaseStyle, ...(isChecked ? checkedStyle : {}), ...(isDisabled ? disabledStyle : {}) }}>
<input type="radio" name={option.name} value={item.id} checked={isChecked} disabled={isDisabled} onChange={() => !isDisabled && handleRadioChange(option.name, item.id)} style={{ display: 'none' }} />
{item.label}
{item.subtitle && <small style={{ ...subtitleStyle, color: isChecked ? 'rgba(255,255,255,0.85)' : 'inherit' }}>{item.subtitle}</small>}
</label>
@@ -8,15 +8,16 @@ export const Llama4MaverickDeployment = () => {
{ id: 'h200', label: 'H200', default: false },
{ id: 'mi300x', label: 'MI300x', default: true },
{ id: 'mi325x', label: 'MI325x', default: false },
{ id: 'mi355x', label: 'MI355x', default: false }
{ id: 'mi355x', label: 'MI355x', default: false },
{ id: 'xeon', label: 'XEON', default: false }
]
},
quantization: {
name: 'quantization',
title: 'Quantization',
items: [
getDynamicItems: (values) => [
{ id: 'bf16', label: 'BF16', default: true },
{ id: 'fp8', label: 'FP8', default: false }
{ id: 'fp8', label: 'FP8', default: false, disabled: values.hardware === 'xeon' }
]
},
toolcall: {
@@ -30,6 +31,7 @@ export const Llama4MaverickDeployment = () => {
speculative: {
name: 'speculative',
title: 'Speculative Decoding (EAGLE3)',
condition: (values) => values.hardware !== 'xeon',
items: [
{ id: 'disabled', label: 'Disabled', default: true },
{ id: 'enabled', label: 'Enable EAGLE3', default: false }
@@ -63,9 +65,11 @@ export const Llama4MaverickDeployment = () => {
cmd += ` \\\n --tp 8`;
} else if (hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi355x') {
cmd += ` \\\n --tp 8`;
} else if (hardware === 'xeon') {
cmd += ` \\\n --device cpu \\\n --disable-overlap-schedule \\\n --tp 6`;
}
if (quantization === 'fp8') {
if (quantization === 'fp8' && hardware !== 'xeon') {
cmd += ` \\\n --quantization fp8`;
}
@@ -73,7 +77,7 @@ export const Llama4MaverickDeployment = () => {
cmd += ` \\\n --tool-call-parser pythonic`;
}
if (speculative === 'enabled') {
if (speculative === 'enabled' && hardware !== 'xeon') {
cmd += ` \\\n --speculative-algorithm EAGLE3 \\\n`;
cmd += ` --speculative-draft-model-path lmsys/sglang-EAGLE3-Llama-4-Maverick-17B-128E-Instruct-v1 \\\n`;
cmd += ` --speculative-num-steps 3 \\\n`;
@@ -151,7 +155,14 @@ export const Llama4MaverickDeployment = () => {
}, []);
const handleRadioChange = (optionName, value) => {
setValues((prev) => ({ ...prev, [optionName]: value }));
setValues((prev) => {
const next = { ...prev, [optionName]: value };
if (optionName === 'hardware' && value === 'xeon') {
next.quantization = 'bf16';
next.speculative = 'disabled';
}
return next;
});
};
const handleCheckboxChange = (optionName, itemId, isChecked) => {
@@ -9,15 +9,16 @@ export const Llama4ScoutDeployment = () => {
{ id: 'h200', label: 'H200', default: false },
{ id: 'mi300x', label: 'MI300x', default: false },
{ id: 'mi325x', label: 'MI325x', default: false },
{ id: 'mi355x', label: 'MI355x', default: false }
{ id: 'mi355x', label: 'MI355x', default: false },
{ id: 'xeon', label: 'XEON', default: false }
]
},
quantization: {
name: 'quantization',
title: 'Quantization',
items: [
getDynamicItems: (values) => [
{ id: 'bf16', label: 'BF16', default: true },
{ id: 'fp8', label: 'FP8', default: false }
{ id: 'fp8', label: 'FP8', default: false, disabled: values.hardware === 'xeon' }
]
},
toolcall: {
@@ -31,6 +32,7 @@ export const Llama4ScoutDeployment = () => {
speculative: {
name: 'speculative',
title: 'Speculative Decoding (EAGLE3)',
condition: (values) => values.hardware !== 'xeon',
items: [
{ id: 'disabled', label: 'Disabled', default: true },
{ id: 'enabled', label: 'Enable EAGLE3', default: false }
@@ -64,9 +66,11 @@ export const Llama4ScoutDeployment = () => {
cmd += ` \\\n --tp 8`;
} else if (hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi355x') {
cmd += ` \\\n --tp 8`;
} else if (hardware === 'xeon') {
cmd += ` \\\n --device cpu \\\n --disable-overlap-schedule \\\n --tp 6`;
}
if (quantization === 'fp8') {
if (quantization === 'fp8' && hardware !== 'xeon') {
cmd += ` \\\n --quantization fp8`;
}
@@ -74,7 +78,7 @@ export const Llama4ScoutDeployment = () => {
cmd += ` \\\n --tool-call-parser pythonic`;
}
if (speculative === 'enabled') {
if (speculative === 'enabled' && hardware !== 'xeon') {
cmd += ` \\\n --speculative-algorithm EAGLE3 \\\n`;
cmd += ` --speculative-draft-model-path lmsys/sglang-EAGLE3-Llama-4-Scout-17B-16E-Instruct-v1 \\\n`;
cmd += ` --speculative-num-steps 3 \\\n`;
@@ -152,7 +156,14 @@ export const Llama4ScoutDeployment = () => {
}, []);
const handleRadioChange = (optionName, value) => {
setValues((prev) => ({ ...prev, [optionName]: value }));
setValues((prev) => {
const next = { ...prev, [optionName]: value };
if (optionName === 'hardware' && value === 'xeon') {
next.quantization = 'bf16';
next.speculative = 'disabled';
}
return next;
});
};
const handleCheckboxChange = (optionName, itemId, isChecked) => {
@@ -6,7 +6,8 @@ export const Qwen25VLDeployment = () => {
items: [
{ id: 'mi300x', label: 'MI300X', default: true },
{ id: 'mi325x', label: 'MI325X', default: false },
{ id: 'mi355x', label: 'MI355X', default: false }
{ id: 'mi355x', label: 'MI355X', default: false },
{ id: 'xeon', label: 'XEON', default: false }
]
},
modelsize: {
@@ -33,25 +34,29 @@ export const Qwen25VLDeployment = () => {
baseName: '72B',
mi300x: { tp: 8, ep: 0 },
mi325x: { tp: 8, ep: 0 },
mi355x: { tp: 8, ep: 0 }
mi355x: { tp: 8, ep: 0 },
xeon: { tp: 6, ep: 0 }
},
'32b': {
baseName: '32B',
mi300x: { tp: 2, ep: 0 },
mi325x: { tp: 2, ep: 0 },
mi355x: { tp: 2, ep: 0 }
mi355x: { tp: 2, ep: 0 },
xeon: { tp: 6, ep: 0 }
},
'7b': {
baseName: '7B',
mi300x: { tp: 1, ep: 0 },
mi325x: { tp: 1, ep: 0 },
mi355x: { tp: 1, ep: 0 }
mi355x: { tp: 1, ep: 0 },
xeon: { tp: 3, ep: 0 }
},
'3b': {
baseName: '3B',
mi300x: { tp: 1, ep: 0 },
mi325x: { tp: 1, ep: 0 },
mi355x: { tp: 1, ep: 0 }
mi355x: { tp: 1, ep: 0 },
xeon: { tp: 3, ep: 0 }
}
};
@@ -73,6 +78,10 @@ export const Qwen25VLDeployment = () => {
let cmd = 'python -m sglang.launch_server \\\n';
cmd += ` --model ${modelName}`;
if (hardware === 'xeon') {
cmd += ` \\\n --device cpu \\\n --disable-overlap-schedule`;
}
if (hwConfig.tp > 1) {
cmd += ` \\\n --tp ${hwConfig.tp}`;
}