docs: sync LMSYS SGLang blog cards (#35416)
Co-authored-by: sglang-bot <sglang-bot@users.noreply.github.com>
This commit is contained in:
+30
-30
@@ -90,7 +90,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
}}
|
||||
>
|
||||
<a
|
||||
href="https://lmsys.org/blog/2026-08-17-advanced-cuda-graph/"
|
||||
href="https://lmsys.org/blog/2026-08-26-qwen-flash-next/"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{
|
||||
@@ -111,8 +111,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
}}
|
||||
>
|
||||
<img
|
||||
src="https://lmsys.org/images/blog/breakable_cuda_graph/bcg-design.svg"
|
||||
alt="Advanced CUDA Graph Techniques in SGLang"
|
||||
src="https://lmsys.org/images/blog/qwen-flash-next/cover.png"
|
||||
alt="Qwen3.8-Flash-Next: Day-0 Support in SGLang"
|
||||
style={{
|
||||
width: "100%",
|
||||
height: "100%",
|
||||
@@ -131,7 +131,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
fontSize: "0.98rem",
|
||||
}}
|
||||
>
|
||||
{"Advanced CUDA Graph Techniques in SGLang"}
|
||||
{"Qwen3.8-Flash-Next: Day-0 Support in SGLang"}
|
||||
</p>
|
||||
<p
|
||||
style={{
|
||||
@@ -140,12 +140,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
opacity: 0.75,
|
||||
}}
|
||||
>
|
||||
{"August 17, 2026"}
|
||||
{"August 26, 2026"}
|
||||
</p>
|
||||
</div>
|
||||
</a>
|
||||
<a
|
||||
href="https://lmsys.org/blog/2026-08-12-qwen3-8-day0-support/"
|
||||
href="https://lmsys.org/blog/2026-08-21-sglang-fast-recovery/"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{
|
||||
@@ -166,8 +166,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
}}
|
||||
>
|
||||
<img
|
||||
src="https://lmsys.org/images/blog/qwen3-8-day0-support/cover-qwen3-8.png"
|
||||
alt="SGLang and Miles Add Day-0 Support for Qwen3.8"
|
||||
src="https://lmsys.org/images/blog/sglang-fast-recovery/preview.png"
|
||||
alt="Fast Engine Recovery: Sub-Second Engine Restart for SGLang via Weight Cache Daemon"
|
||||
style={{
|
||||
width: "100%",
|
||||
height: "100%",
|
||||
@@ -186,7 +186,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
fontSize: "0.98rem",
|
||||
}}
|
||||
>
|
||||
{"SGLang and Miles Add Day-0 Support for Qwen3.8"}
|
||||
{"Fast Engine Recovery: Sub-Second Engine Restart for SGLang via Weight Cache Daemon"}
|
||||
</p>
|
||||
<p
|
||||
style={{
|
||||
@@ -195,12 +195,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
opacity: 0.75,
|
||||
}}
|
||||
>
|
||||
{"Aug 12, 2026"}
|
||||
{"August 21, 2026"}
|
||||
</p>
|
||||
</div>
|
||||
</a>
|
||||
<a
|
||||
href="https://lmsys.org/blog/2026-08-11-unified-radix-cache/"
|
||||
href="https://lmsys.org/blog/2026-08-21-ling3-flash-spec-decode-blackwell/"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{
|
||||
@@ -221,8 +221,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
}}
|
||||
>
|
||||
<img
|
||||
src="https://lmsys.org/images/blog/unified-radix-cache/image1.svg"
|
||||
alt="Unified Radix Cache: One Tree for Hybrid Model Prefix Caching"
|
||||
src="https://lmsys.org/images/blog/ling3-flash-batch1/00_headline.png"
|
||||
alt="Chasing the Batch-1 Floor: Ling-3.0-flash Speculative Decode on Blackwell"
|
||||
style={{
|
||||
width: "100%",
|
||||
height: "100%",
|
||||
@@ -241,7 +241,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
fontSize: "0.98rem",
|
||||
}}
|
||||
>
|
||||
{"Unified Radix Cache: One Tree for Hybrid Model Prefix Caching"}
|
||||
{"Chasing the Batch-1 Floor: Ling-3.0-flash Speculative Decode on Blackwell"}
|
||||
</p>
|
||||
<p
|
||||
style={{
|
||||
@@ -250,12 +250,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
opacity: 0.75,
|
||||
}}
|
||||
>
|
||||
{"August 11, 2026"}
|
||||
{"August 21, 2026"}
|
||||
</p>
|
||||
</div>
|
||||
</a>
|
||||
<a
|
||||
href="https://lmsys.org/blog/2026-08-11-nemotron-3-5-lightning/"
|
||||
href="https://lmsys.org/blog/2026-08-20-miles-mooncake-rollout-data-transfer/"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{
|
||||
@@ -276,8 +276,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
}}
|
||||
>
|
||||
<img
|
||||
src="https://lmsys.org/images/blog/nemotron-3-5-lightning/pinchbench-accuracy-vs-time.png"
|
||||
alt="SGLang Adds Day-0 Support for NVIDIA Nemotron 3.5 Lightning"
|
||||
src="https://lmsys.org/images/blog/miles-mooncake-rollout-data-transfer/featured.png"
|
||||
alt="Mooncake for Miles: From Fragmented Rollout Data to Efficient Bulk I/O"
|
||||
style={{
|
||||
width: "100%",
|
||||
height: "100%",
|
||||
@@ -296,7 +296,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
fontSize: "0.98rem",
|
||||
}}
|
||||
>
|
||||
{"SGLang Adds Day-0 Support for NVIDIA Nemotron 3.5 Lightning"}
|
||||
{"Mooncake for Miles: From Fragmented Rollout Data to Efficient Bulk I/O"}
|
||||
</p>
|
||||
<p
|
||||
style={{
|
||||
@@ -305,12 +305,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
opacity: 0.75,
|
||||
}}
|
||||
>
|
||||
{"August 11, 2026"}
|
||||
{"August 20, 2026"}
|
||||
</p>
|
||||
</div>
|
||||
</a>
|
||||
<a
|
||||
href="https://lmsys.org/blog/2026-08-10-meta-muse-glimmer/"
|
||||
href="https://lmsys.org/blog/2026-08-19-deepseek-v4-pro-engine-optimization-h20/"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{
|
||||
@@ -331,8 +331,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
}}
|
||||
>
|
||||
<img
|
||||
src="https://lmsys.org/images/blog/2026-08-10-meta-muse-glimmer/cover-muse-glimmer.png"
|
||||
alt="SGLang Adds Day-0 Support for Muse Glimmer, a Multimodal Model Built for Local Agentic Workflows"
|
||||
src="https://lmsys.org/images/blog/deepseek_v4/00_cover.png"
|
||||
alt="Pushing the Limits of Serving DeepSeek-V4-Pro"
|
||||
style={{
|
||||
width: "100%",
|
||||
height: "100%",
|
||||
@@ -351,7 +351,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
fontSize: "0.98rem",
|
||||
}}
|
||||
>
|
||||
{"SGLang Adds Day-0 Support for Muse Glimmer, a Multimodal Model Built for Local Agentic Workflows"}
|
||||
{"Pushing the Limits of Serving DeepSeek-V4-Pro"}
|
||||
</p>
|
||||
<p
|
||||
style={{
|
||||
@@ -360,12 +360,12 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
opacity: 0.75,
|
||||
}}
|
||||
>
|
||||
{"August 10, 2026"}
|
||||
{"August 19, 2026"}
|
||||
</p>
|
||||
</div>
|
||||
</a>
|
||||
<a
|
||||
href="https://lmsys.org/blog/2026-08-07-hpc-ops-sglang/"
|
||||
href="https://lmsys.org/blog/2026-08-18-miles-v0-1/"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{
|
||||
@@ -386,8 +386,8 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
}}
|
||||
>
|
||||
<img
|
||||
src="https://lmsys.org/images/blog/hpc-ops-sglang/hpc-ops-sglang-cover.webp"
|
||||
alt="HPC-Ops \u00d7 SGLang: High-Performance Attention, Router GEMM, and MoE Kernels from Tencent Hunyuan"
|
||||
src="https://lmsys.org/images/blog/miles-v0-1/miles-logo.png"
|
||||
alt="Miles v0.1: Production-level Post-training"
|
||||
style={{
|
||||
width: "100%",
|
||||
height: "100%",
|
||||
@@ -406,7 +406,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
fontSize: "0.98rem",
|
||||
}}
|
||||
>
|
||||
{"HPC-Ops \u00d7 SGLang: High-Performance Attention, Router GEMM, and MoE Kernels from Tencent Hunyuan"}
|
||||
{"Miles v0.1: Production-level Post-training"}
|
||||
</p>
|
||||
<p
|
||||
style={{
|
||||
@@ -415,7 +415,7 @@ It is designed to deliver low-latency and high-throughput inference across a wid
|
||||
opacity: 0.75,
|
||||
}}
|
||||
>
|
||||
{"August 7, 2026"}
|
||||
{"August 18, 2026"}
|
||||
</p>
|
||||
</div>
|
||||
</a>
|
||||
|
||||
Reference in New Issue
Block a user