docs: sync LMSYS SGLang blog cards (#32982)
Co-authored-by: sglang-bot <sglang-bot@users.noreply.github.com>
This commit is contained in:
+30
-30
@@ -90,7 +90,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<a
|
<a
|
||||||
href="https://lmsys.org/blog/2026-07-29-mxfp8-nvfp4-rl/"
|
href="https://lmsys.org/blog/2026-08-12-qwen3-8-day0-support/"
|
||||||
target="_blank"
|
target="_blank"
|
||||||
rel="noopener noreferrer"
|
rel="noopener noreferrer"
|
||||||
style={{
|
style={{
|
||||||
@@ -111,8 +111,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<img
|
<img
|
||||||
src="https://lmsys.org/images/blog/mxfp8-nvfp4-rl/NVFP4-dq.drawio.png"
|
src="https://lmsys.org/images/blog/qwen3-8-day0-support/cover-qwen3-8.png"
|
||||||
alt="Towards Blackwell-Native 8-bit and 4-bit RL: End-to-End MXFP8 and NVFP4 RL in Miles"
|
alt="SGLang and Miles Add Day-0 Support for Qwen3.8"
|
||||||
style={{
|
style={{
|
||||||
width: "100%",
|
width: "100%",
|
||||||
height: "100%",
|
height: "100%",
|
||||||
@@ -131,7 +131,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
fontSize: "0.98rem",
|
fontSize: "0.98rem",
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"Towards Blackwell-Native 8-bit and 4-bit RL: End-to-End MXFP8 and NVFP4 RL in Miles"}
|
{"SGLang and Miles Add Day-0 Support for Qwen3.8"}
|
||||||
</p>
|
</p>
|
||||||
<p
|
<p
|
||||||
style={{
|
style={{
|
||||||
@@ -140,12 +140,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
opacity: 0.75,
|
opacity: 0.75,
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"July 29, 2026"}
|
{"Aug 12, 2026"}
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
</a>
|
</a>
|
||||||
<a
|
<a
|
||||||
href="https://lmsys.org/blog/2026-07-27-kimi-k3-day0-support/"
|
href="https://lmsys.org/blog/2026-08-11-unified-radix-cache/"
|
||||||
target="_blank"
|
target="_blank"
|
||||||
rel="noopener noreferrer"
|
rel="noopener noreferrer"
|
||||||
style={{
|
style={{
|
||||||
@@ -166,8 +166,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<img
|
<img
|
||||||
src="https://lmsys.org/images/blog/kimi-k3-day0-support/cover-kimi-k3.png"
|
src="https://lmsys.org/images/blog/unified-radix-cache/image1.svg"
|
||||||
alt="SGLang and Miles Add Day-0 Support for Kimi K3"
|
alt="Unified Radix Cache: One Tree for Hybrid Model Prefix Caching"
|
||||||
style={{
|
style={{
|
||||||
width: "100%",
|
width: "100%",
|
||||||
height: "100%",
|
height: "100%",
|
||||||
@@ -186,7 +186,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
fontSize: "0.98rem",
|
fontSize: "0.98rem",
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"SGLang and Miles Add Day-0 Support for Kimi K3"}
|
{"Unified Radix Cache: One Tree for Hybrid Model Prefix Caching"}
|
||||||
</p>
|
</p>
|
||||||
<p
|
<p
|
||||||
style={{
|
style={{
|
||||||
@@ -195,12 +195,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
opacity: 0.75,
|
opacity: 0.75,
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"July 27, 2026"}
|
{"August 11, 2026"}
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
</a>
|
</a>
|
||||||
<a
|
<a
|
||||||
href="https://lmsys.org/blog/2026-07-18-opd-support-in-miles/"
|
href="https://lmsys.org/blog/2026-08-11-nemotron-3-5-lightning/"
|
||||||
target="_blank"
|
target="_blank"
|
||||||
rel="noopener noreferrer"
|
rel="noopener noreferrer"
|
||||||
style={{
|
style={{
|
||||||
@@ -221,8 +221,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<img
|
<img
|
||||||
src="https://lmsys.org/images/blog/opd-support-in-miles/figure-1.png"
|
src="https://lmsys.org/images/blog/nemotron-3-5-lightning/pinchbench-accuracy-vs-time.png"
|
||||||
alt="OPD Support in Miles"
|
alt="SGLang Adds Day-0 Support for NVIDIA Nemotron 3.5 Lightning"
|
||||||
style={{
|
style={{
|
||||||
width: "100%",
|
width: "100%",
|
||||||
height: "100%",
|
height: "100%",
|
||||||
@@ -241,7 +241,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
fontSize: "0.98rem",
|
fontSize: "0.98rem",
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"OPD Support in Miles"}
|
{"SGLang Adds Day-0 Support for NVIDIA Nemotron 3.5 Lightning"}
|
||||||
</p>
|
</p>
|
||||||
<p
|
<p
|
||||||
style={{
|
style={{
|
||||||
@@ -250,12 +250,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
opacity: 0.75,
|
opacity: 0.75,
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"July 18, 2026"}
|
{"August 11, 2026"}
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
</a>
|
</a>
|
||||||
<a
|
<a
|
||||||
href="https://lmsys.org/blog/2026-07-15-inkling-day0-support/"
|
href="https://lmsys.org/blog/2026-08-10-meta-muse-glimmer/"
|
||||||
target="_blank"
|
target="_blank"
|
||||||
rel="noopener noreferrer"
|
rel="noopener noreferrer"
|
||||||
style={{
|
style={{
|
||||||
@@ -276,8 +276,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<img
|
<img
|
||||||
src="https://lmsys.org/images/blog/inkling-day0-support/inkling-cover.png"
|
src="https://lmsys.org/images/blog/2026-08-10-meta-muse-glimmer/cover-muse-glimmer.png"
|
||||||
alt="SGLang and Miles Add Day-0 Support for Inkling, a Frontier Multimodal Model"
|
alt="SGLang Adds Day-0 Support for Muse Glimmer, a Multimodal Model Built for Local Agentic Workflows"
|
||||||
style={{
|
style={{
|
||||||
width: "100%",
|
width: "100%",
|
||||||
height: "100%",
|
height: "100%",
|
||||||
@@ -296,7 +296,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
fontSize: "0.98rem",
|
fontSize: "0.98rem",
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"SGLang and Miles Add Day-0 Support for Inkling, a Frontier Multimodal Model"}
|
{"SGLang Adds Day-0 Support for Muse Glimmer, a Multimodal Model Built for Local Agentic Workflows"}
|
||||||
</p>
|
</p>
|
||||||
<p
|
<p
|
||||||
style={{
|
style={{
|
||||||
@@ -305,12 +305,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
opacity: 0.75,
|
opacity: 0.75,
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"July 15, 2026"}
|
{"August 10, 2026"}
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
</a>
|
</a>
|
||||||
<a
|
<a
|
||||||
href="https://lmsys.org/blog/2026-07-13-glm52-optimization/"
|
href="https://lmsys.org/blog/2026-08-07-hpc-ops-sglang/"
|
||||||
target="_blank"
|
target="_blank"
|
||||||
rel="noopener noreferrer"
|
rel="noopener noreferrer"
|
||||||
style={{
|
style={{
|
||||||
@@ -331,8 +331,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<img
|
<img
|
||||||
src="https://lmsys.org/images/blog/glm52-optimization/glm52-nvfp4-performance-pareto.png"
|
src="https://lmsys.org/images/blog/hpc-ops-sglang/hpc-ops-sglang-cover.webp"
|
||||||
alt="Serving GLM5.2 NVFP4 Agentic Workload with SGLang: Reaching 500 TPS in 2 Weeks"
|
alt="HPC-Ops \u00d7 SGLang: High-Performance Attention, Router GEMM, and MoE Kernels from Tencent Hunyuan"
|
||||||
style={{
|
style={{
|
||||||
width: "100%",
|
width: "100%",
|
||||||
height: "100%",
|
height: "100%",
|
||||||
@@ -351,7 +351,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
fontSize: "0.98rem",
|
fontSize: "0.98rem",
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"Serving GLM5.2 NVFP4 Agentic Workload with SGLang: Reaching 500 TPS in 2 Weeks"}
|
{"HPC-Ops \u00d7 SGLang: High-Performance Attention, Router GEMM, and MoE Kernels from Tencent Hunyuan"}
|
||||||
</p>
|
</p>
|
||||||
<p
|
<p
|
||||||
style={{
|
style={{
|
||||||
@@ -360,12 +360,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
opacity: 0.75,
|
opacity: 0.75,
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"July 14, 2026"}
|
{"August 7, 2026"}
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
</a>
|
</a>
|
||||||
<a
|
<a
|
||||||
href="https://lmsys.org/blog/2026-07-10-rocm-miles-dsv4/"
|
href="https://lmsys.org/blog/2026-08-05-glmImage-optimization/"
|
||||||
target="_blank"
|
target="_blank"
|
||||||
rel="noopener noreferrer"
|
rel="noopener noreferrer"
|
||||||
style={{
|
style={{
|
||||||
@@ -386,8 +386,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<img
|
<img
|
||||||
src="https://lmsys.org/images/blog/rocm-miles-dsv4/preview.png"
|
src="https://lmsys.org/images/blog/2026-08-05-glmImage-optimization/05-fanout.png"
|
||||||
alt="Bringing DeepSeek-V4 Flash RL Training to AMD Instinct MI355X GPUs with Miles"
|
alt="Full-Stack Performance Optimization of AR+DiT in SGL-Diffusion"
|
||||||
style={{
|
style={{
|
||||||
width: "100%",
|
width: "100%",
|
||||||
height: "100%",
|
height: "100%",
|
||||||
@@ -406,7 +406,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
fontSize: "0.98rem",
|
fontSize: "0.98rem",
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"Bringing DeepSeek-V4 Flash RL Training to AMD Instinct MI355X GPUs with Miles"}
|
{"Full-Stack Performance Optimization of AR+DiT in SGL-Diffusion"}
|
||||||
</p>
|
</p>
|
||||||
<p
|
<p
|
||||||
style={{
|
style={{
|
||||||
@@ -415,7 +415,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
|||||||
opacity: 0.75,
|
opacity: 0.75,
|
||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
{"July 10, 2026"}
|
{"August 05, 2026"}
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
</a>
|
</a>
|
||||||
|
|||||||
Reference in New Issue
Block a user