From 40894be3c32dec8e3aca003b8fc00441433fabf7 Mon Sep 17 00:00:00 2001 From: Yi Zhong <207368749+vincentzed@users.noreply.github.com> Date: Thu, 11 Jun 2026 20:08:19 -0700 Subject: [PATCH] add lfm2.5 to new cookbook. (#27409) --- docs_new/cards/logos/liquidai.png | Bin 0 -> 8889 bytes .../autoregressive/LiquidAI/LFM2.5.mdx | 334 ++++++++++++ docs_new/cookbook/autoregressive/intro.mdx | 6 + docs_new/docs.json | 6 + .../configs/LiquidAI/lfm2.5-benchmarks.jsx | 227 +++++++++ .../src/snippets/configs/LiquidAI/lfm2.5.jsx | 477 ++++++++++++++++++ 6 files changed, 1050 insertions(+) create mode 100644 docs_new/cards/logos/liquidai.png create mode 100644 docs_new/cookbook/autoregressive/LiquidAI/LFM2.5.mdx create mode 100644 docs_new/src/snippets/configs/LiquidAI/lfm2.5-benchmarks.jsx create mode 100644 docs_new/src/snippets/configs/LiquidAI/lfm2.5.jsx diff --git a/docs_new/cards/logos/liquidai.png b/docs_new/cards/logos/liquidai.png new file mode 100644 index 0000000000000000000000000000000000000000..6c06782fb7190f5605bebfbfc0cdb767e400b83e GIT binary patch literal 8889 zcmeHs`9IX(7r&)avNZ3S$XerF-Wfy)*-9hri)E~1{X~{YjD0W3RH)uHC|fEaGh>~x zGubM8maz=R5`)2DjIsM(eZJqv_n-LqZSL#bbI(2JdCqyB>)D-K#=?RUf;>Dt!dI^# zOn7+s@jN_y!u$7umGeSN6+Aq!$g7BpW`QYlL!vF2=Bcs^vTAigbqij<^RO50S{=;Y zKg3sCxMY01$2@?57ypd-+i<+l{T25075FpU1&s%TSB5Uc?inAGc=p=T(ur#FSar{} z{TYW659wsysqLJe)k+QHPN3t{iZ&Kj+ct(n=pih2N{B)F@89t*!`!fWZP%o=VN4qf z0Sx-z-~T-De?351p3CgwbqJ-#Er#$?DmShmQieH~z{d++1(1RxklR*ndwF<1+_rgC zgX3o!B5VgZ{Jh|+&n?7kEOtn{xcUgh#!nb*Kn>`?WLhrzS=c$=tf{_A*bmmqmlv^? zEslOUi_9aox6i7BCEw8wDU$%khw|=K`ZQ<0lJ6i`G_8`DXA&X@JayR2nAihgrOVd@ zozFD&f$zo{M7?9ODSlt z7!ZBB^ox3KIP<_0YKg?*-QNX%apsQ%`Xy5A(qpllKbN{Vi(_3!?`yvLlnNKxbyrlW z!{ZZvWeH>{S%+Q4mRHMVIg6Fhz<+adTg|E6^D2s975{tSfZ1Ld_2RTYr(3q2Sd&P; zdep|6h)d^OKzRNMhdQC1=F|MH(=E*rz1;#A-c^4y>)hjU^){k%Nz?oA`4y~I)24c+ky2&OhHWLU^ib0~1_q6x^3BjWYaP!?&*VN(%r0{FZ*x zj!nRPB;ku?+E+0whM~=lH{0k3rD~ z?OMOJ79H&$L?zf0!6x@}zF$Xl~w10@tDxK_%+R?dcijq9xEjX~XiS)Y+z z@OVl5eT!PyWNpY?J3Ma+K2oQ$v4KPgQq1SNd8EDC2zS-fxp~f{-KOl)%A-&bfYC^q zYc!{tl&HZrWcKwuP2>7<-s<+o=U?}p(MaRC+e86ZtxvZ|4G%bCR0aB^A>NyjbmmZ8 z9%?2C&SFhqwR~&Cgn$=bo%v&X5;Jyxh;Fr|X?qk3d-M6Vd{XRK_-P?1>`ntW)`ZjQoc#%irAy~Wkf@|u~{wNJk zVdbr*TSWjx4&|L;pV~>tBQe6hIeOv{roiUA0O32&O3x5<$3H;|?$tU2*IY>8n@rVy zH79c^`?|%L(AWaD&+eB3bN&Qf7AW^v*3rfiNIqWL8|y8Rdz1dn!#kP>3=mR!ZsvV| z?*}B_;O51)-0bwEzEcN%YG%ZIj7EB9B_TE0|FL>t9Jh_Wkw=$&2tAvcQC0J?2>(KR#2__q z3RvpUGXHL*$>jh+OlVcn%QFl`x0xrmmx3<+m(Zb4MrAh`m;H%sfXh_X#erhp!Fu@_Emf^;) z&#=-q>n&8WfT~6EqK3RxKNXs+v$0x%6xV99z(wCAvRsYb zeIx*f(4@1r7B=Gb31Psxye1CFjKAy{r^n-Ck1rKxvNqCn6R) zx)y{v+Kju~M47(M2qm+rmd5VoaP*Y8GU+d|TQ_cj`$hN!z{OVzaB4XM!dee4NMD7y zRX8qJ<1pwnBfYoKM^4XRGCnSOp0*VodxZ@wczk>~G~z{2=0zb0duP3VV&7ZiwXA0o6Tn}4w;Ha z00ZpKpM|A0yXtOKLQDIbFx&s3JpcAHh>%A(k^e>X5J+Wyd0kLLd4lCpTRzwC89jTS z!NH%PGw1XI=N+xB7OUHa0F9yC;3?S0tAxGWMjg3a_zj#6AXf06=nS(}ye; zHM$R(oV-XdQ`Ps54SP*2TMNEtvHsS(LKAi_zTgZv>pV9{D%8Ea@ry2VE=Y&1C5>yZ zPry+#1+DvQH#1bZfQMah6?uV01YeCBPL_J$&$;%g0tLnA<& zC9H#}w{Pcrtu#Lh%=w7!ymyd&^YH9$Ezi*^tIfBr9q%%=yc%i|%?K9r(Fv~&+szi$ zWS@*o6coKzIm0u6=&mcwwJ=i+Jk7lsJe(d|;ByuP!VG0QI^%u-|8mm3y6d5d=rB8< z5Ie1t)olz8cgFk&QE7SiVuruYy(Y{kJPtto7tUapz?sg&!Py#RSppgEZx1Ed)%1Pq z^ZIEJ5)i$`$rM=!q{HEF*J8e#N@$A*8v*L`vUys?)ud1Qz%20@iDU*`2!HbL@Y1@1 z;Uo}3qd0d#{L34QEx>D(EBZ)BOVp1Os!;2Jl5uYXQmvu+gms$1OeqW4v@(^1?YrAw z^)Y9zC6BS};Mt|2N+&25C-jVn!Nw>=okM_!pizj>DyW0|{gX$+b-Y~Xqcn}`uLt#% zB#hWMZeHGPaFNy8Y6AE|YpY8(Ef<*+*uYE6&8;}7gF^4%XMd;ORDj_Q#iPCv_jq&H z@DT%b8M^8)vN^e`e3^j?@=ybTW9VD;ziZCCy#GeV!pKxsb4bVSI&xv{_BK%mv@Zt_ zXFqg7A?1DA+E*D-N9)W`a&0)OMRw0@ub%)w5F0*J5oD|{U~QU_nYmd>WBX6)>r@DY zXaj;GY@E}VBGJG+^oAWXxNB%|rG7~yS-3|LA%j}SwLuJ{;TB;5gBwMJsp4Vo>$Hy zzGO0A^PpYw_m6Aa>%&Kv1f*&IFVacUAx8DBP$Drwr9=-lEM7EGF0;(aC$ywWvQvPY zEsFwvFIbpbqW4jgU6yqho1uh!Ws-taddHtAaBO5D?InB;N=kaH>yQEsm#%)jsqwpw ze^==zGQikC`Jr#+W~nq+drW4dn6UDzNnxKRqw{;k`~?6qI|0t|!B`2b$?Np2$r1c- zFLO)P(&FNV%~yXwn0@II+ExvpV*|5m+4y@#n@fLApQ@>o8|%M3`f=cK`z1`JbwGHutCg5Uuc-8| zKR4e}E?!QR`rBPl@8svq1$J7Jy(Da`BJOyp*;P1Zt{LMyFAg%b^D6XdkBb(lXe(*# zyAhZ4zgu_pk58uPl>G_gx~nNd_kH^#>Jk5Ij@fgN#RQ^Gf&6yyFHMcOaiXpOW9t&g zqYo2N1RQZ-pd@z608D0fv)_`0Xn#hx@VdJ}(S>`S4h*zuD;F2bi*X%nF9X0z_x;z# zhqpx^H(3uDSIkpntYsm$%>ol2xCOu|06{4@e#5v77u7uCd3I@~sgFSq*~e~j?AUf7 z9}guouNIxFo}X}N#TXUfw0FFsqHs~Eecck_Ia_`hfb@Z_)pWwI2B})Rcx>2>G$_i_ zO_gv~-oQDOm&ajSDiutn$jb8I7nu0m0nPM#sBF&^F}9k4BqT4jQ5H%MdS8SmqzR7 zZ7%?(lH{F(ZQas7q_}+itq|?n)OujAwAE`R$f850vt1h3EntqXMV96Pay=bU9|T$4 zfjM33X5cp0-^8bXI(7OtJvm}a^lqZMal|o**K;h_Z%YNh^|vP)Z{0(NmX-~G`5`dw z+1a-*6bMHLWj5;KbSRtowSp=JE6=ez-Ik6v7K<}8_7j5Nh30FF>gf|6k=G2U|3QV% zn;iq(JV{l468crb0!d?7!Heyc&CJ^rjx5_;#;OpXaHgNkIa@=Jp>E z8zp{QJkWu6EOo-7lt+QG!G=Aeew`@0FY;${bVMUq6g{Fj&0fwP=bFdL2Wr$P9qoU0Rr_2s$!XC6U^j+`yX)O6W zJn~boP*5%{9#rrMK9l|e8;3GD#PdcKz7xnh($jhFX9PgW)>iCRRJh~iwsHSaFG8q_a6wN^kZWPRApoJ+=Po%G^Qi=LYXS&<~A8J&N28j0M;YVW0~l)jOq z)r!eiJ|4i&hqxjtmGTzK4FHqs`*fz{u|1mbgdn|EDHpX7=wZXrGIY0Ev~v$-xw8G~ z<4aBYK;4B-WJ@BZBik?2&p2GKJxSCLPLcLdEH#_If)*H7Pn5x8&rEn=o|pj^@E;a> zPh6&;%>i5bq4<6N_Ys~pj?DSvCrEns;o=YgtCG@FfGvn87W4Ug$;J9gU+68A<{a;G zI?f;?vn)I^BWUb#B6ighv57=@zBi$JwYfUbXN45OqeP4*qJBl{iu<61Ejcr-%g-3O z#M@9}liLLkJJhkQM}LL|xyi(geg^MsmuBRB9ubzMzTcX51q?!O;T^Bx4SSoQNm)xj zo1VKNS|Q~oq>+lHW1~?^YrSrL2KGVh=+IN-HQmIU)5k!+Nz*zGOS_N9_lR3}VPwH! zD}8oxO&t};W6X`EYcu+$n}-~(BWz#sFBK6)Zy)vf5&jQsCAAaK7#=-EirF}MWT}PD zEI%CL$rLotNN>Kl`0@0<+A5#S8MP%OeL^roN&WeyO#j-qX<$&Adgm9%cS9qF>U%1Y zzWfXsv88BXv3)@#zUx(6m5dZ5BB|sd%S|eoowvDOu@4H0Eo~Hg*#-->ZSmfhCZXfHYEa1yzQq`y){? z9d~QnoQC%`2UeYe4NF6~hvK*Lc6P297NsyxS7b!u!4H28(h{2MUuZ*!H7L7HmOMnbI zbnxAnQ1}->H|PQ5M51WUbP9_ES?oLO7}3FWZ46q`LhnWHo-1C;(0>W2*leB5ekQ; z*?qAu7tEQAov$I-))bUdgSFrw^e5jsu~R5;C&h9o@=^Fct~X7&rpi zQG^lOWKYdbSFHq9Hl^me(RQbxAD#W}2F<+L)UV%~3FP5_+o!1xJJ>)Qge?TpuFMW! zm%5dBFn5|GS~0(y)#s#MMg3CG(|vbY^-K3~&v+3b#JE07k~?+|zCqPGIw3Vz2Pptt z9bCrcdqeY+s`oVF<&larF#T8jPU*ytWw}t3w&M!#Q`I`@ndt^}Z#pQdYBR-B=iLY6%S~+VN7wplPHDo% zl0=@XWfpzak=W9OZzbEM|Ac=B=soPsjztUGDO-{(_SGDPc(pgYq)r%`jVDwyPR6C{ zc*ySmys$NVU|Et!&GWB_Q9cz~e2YBOg;# zBR$Fh(>YUajNa;yzI_SdiN)^x1z}7c2k&Pps)^ z4qHuE96qHk2F4J(ljdDvPHl!vpHa@lCev7MYzi{@gXv=cC~}X#eP*xZ)*2A|GF>mT z=cBTXm2j0NOhhJQ=T>V&s*46bVZ|0PtNCaRWL^G8nl;>w-SM8qnS`(&R24%4UA-kA zvcpg#?O+P*hz?6A`^&|yb|&lcKXysaMThclWG9o5!@U;JKtLhOm-8kaY&_Uo=$F_c zTR#nrZ-4gnx?0!K3W54Q7I$@HP0$ODVp3WOmAuU_vz57FtkUHw_Br4 z&`bM2c|P^&VB&#L5@{0i(!};O3d0_{jLjvi10UmX$uK)RJ$?$Y49pBWz#96K z0cmktYP3pv_#b8SHbR>X9NxuBp$vjkl1h6yD&06-hpXJ!2xPb`neESTxu0BueC{5m z=Jf)UsqD@ky#oQQxM4F}H}-zkgs24lQ(kUkm!gT^?4C*R3PCYDsG`?z`Hs1SzqihD zHMWG_pe5(IMUykN(G^P@HRrfM=t9q}msn|OsJsHr!qlPae*M_Bbgw=1Ad%b&{+Jj{ z4M-(3FO)vZ(vV)vPrVLVq{-=Wso1~|F^q>5LU^}m4(f1{9FR)YR)mHBDy_hDj(*)U z{v9@kg`{z(uz?86t?%2mo+&%#yCpM8yO&hk{<&^uPVT71dGNm~FG6P<20l*ms9ykN zpd&8E$ z-8thUY`fQZ0rsJ9g?Biy%p-NF^G`YXZV? zt&gv;jydMK3pSOg8@|ggXW*D2msi2m!sBr2bAFI1{BJ_{dw2mOKG1(9q;}>4C{A+} zulJNBV6fRZ7CdG`5|A{$O*4d)Ms2&cGK}26uW}}aZz3v6QeK4$^z@Xe$Mi&-*Up4N zygT;OcUlS1dg4I|4hg3kcow#jPSM&`C7%l4Vp7+-|N7vsiQIguce_kkOnBz}ero46 zdQF7kIAGX!GE0hUO^bqA#>W2;i}hg-e%Q0Zd%p2Ahst)JpzfEFpeiNC_1bdFcgo*z z+S%MI4N;!_GNCw*irQa*4QQet9N7vw1&cqH@kZn&))i) z+W8E!XaaxY7RC7{58uo0jzJUmy8ZXt+<4v+s2G+OJ+ literal 0 HcmV?d00001 diff --git a/docs_new/cookbook/autoregressive/LiquidAI/LFM2.5.mdx b/docs_new/cookbook/autoregressive/LiquidAI/LFM2.5.mdx new file mode 100644 index 000000000..54270885a --- /dev/null +++ b/docs_new/cookbook/autoregressive/LiquidAI/LFM2.5.mdx @@ -0,0 +1,334 @@ +--- +title: LFM2.5 +description: "Deploy Liquid AI's LFM2.5 with SGLang — hybrid LIV-convolution + GQA models from 350M to the 8B-A1B MoE, plus LFM2.5-VL vision, with reasoning and Pythonic tool calling." +tag: NEW +--- + +## Deployment + + + + + +For all methods and hardware platforms, see the [official SGLang installation guide](../../../docs/get-started/install). The two paths below match the **Python / Docker** toggle in the command panel. + + + + + +```bash Command +pip install --upgrade pip +pip install uv +uv pip install sglang +``` + + +LFM2.5 support — the dense / MoE / VL model classes and the `lfm2` tool-call parser — ships on SGLang `main`. If your installed release predates it, install from source or use the Docker dev image. + + +Then run the **Python** output of the command panel below in that environment. + + + + + +LFM2.5 support ships in the pinned SGLang dev image: + +```bash Command +docker pull lmsysorg/sglang:dev-cu13 +``` + +For how to launch the image, see [Install → Method 3: Using Docker](../../../docs/get-started/install#method-3-using-docker). A minimal example (substitute the inner `sglang serve ...` with whatever the command generator below produces): + +```bash Command +docker run --gpus all \ + --shm-size 32g \ + -p 30000:30000 \ + -v ~/.cache/huggingface:/root/.cache/huggingface \ + --env "HF_TOKEN=" \ + --ipc=host \ + lmsysorg/sglang:dev-cu13 \ + sglang serve +``` + + + + + + + +Every LFM2.5 model runs on a **single GPU (TP=1)** — pick your hardware + model variant to generate the launch command. One recipe covers all operating points per variant; the commands differ only by the parsers a model needs and, on Blackwell, the attention backend. The `lfm2` tool-call parser and each reasoning model's `--reasoning-parser` are already part of the verified command. + +import { Deployment } from "/src/snippets/_deployment.jsx"; +import { config } from "/src/snippets/configs/LiquidAI/lfm2.5.jsx"; +import { benchmarks } from "/src/snippets/configs/LiquidAI/lfm2.5-benchmarks.jsx"; + + + +
+

Panel controls (top of the command box):

+
+
+ +## Playground + +The Playground is where you experiment with **SGLang features beyond the verified matrix**. The Deploy panel above only emits combinations that have been signed off on; the Playground lets you turn on additional knobs on top of whichever cell the Deploy panel is currently showing. The base is read live from your Deploy selection — only your overrides change. + +For LFM2.5 the exposed knob is the **TP override** (every variant is verified at TP=1; TP=2 is available for experimentation on the larger checkpoints). The reasoning and tool-call parsers are not playground toggles here — they are variant-intrinsic and already baked into each verified command. + +Lines highlighted **green** are added by your overrides; lines with **red strikethrough** were in the verified base but stripped by an override. When no override differs from the base cell, the playground inherits the base's **Verified** badge; any actual change flips it to **Not Verified** until the new configuration is run end-to-end and submitted back. + +import { Playground } from "/src/snippets/_playground.jsx"; + + + +
+

Panel controls reuse Python / Docker · ⧉ Copy · $ cURL · ⚙ Env from the Deploy panel, plus one extra:

+
    +
  • Submit ↗ — opens a pre-filled GitHub issue so you can land your override combo as a new verified cookbook cell. Shown only while the badge says Not Verified; click it once you've actually run the command on your hardware and confirmed it works.
  • +
+
+ +## 1. Model Introduction + +LFM2.5 is [Liquid AI](https://www.liquid.ai/)'s family of hybrid models for on-device deployment, built on the LFM2 architecture with extended pre-training and large-scale reinforcement learning, released under the [LFM Open License v1.0](https://huggingface.co/LiquidAI/LFM2.5-8B-A1B/blob/main/LICENSE). The backbone interleaves **double-gated LIV (linear input-varying) convolution blocks** with a small number of **GQA full-attention blocks**: the convolution blocks give linear-time, low-memory sequence mixing while the periodic attention blocks preserve associative recall. + +**Key Features:** + +- **Hybrid LIV-conv + GQA architecture**: the 1.2B / 350M dense models are 16 layers (10 conv + 6 GQA); the 8B-A1B MoE is 24 layers (18 conv + 6 GQA). +- **Pythonic tool calling**: function calls are emitted as a Python list between `<|tool_call_start|>` and `<|tool_call_end|>` tokens. The `lfm2` tool-call parser surfaces these as standard `message.tool_calls`. +- **Reasoning variants**: the 8B-A1B and 1.2B-Thinking checkpoints emit an explicit `...` chain-of-thought before the answer. +- **Multilingual**: up to 10 languages, with dedicated Japanese chat checkpoints. +- **Vision**: LFM2.5-VL-1.6B pairs the 1.2B language backbone with a SigLIP2 NaFlex 400M encoder for OCR, document understanding, and multilingual vision; LFM2.5-VL-450M pairs the 350M backbone with a SigLIP2 86M encoder for captioning and object detection at edge sizes. + +**Available Models:** + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
ModelParametersContextRole
LFM2.5-8B-A1B8.3B total / 1.5B active (MoE)128KReasoning-tuned, agentic / tool use
LFM2.5-1.2B-Instruct1.17B (dense)32KGeneral instruct, RAG, data extraction
LFM2.5-1.2B-Thinking1.17B (dense)32KReasoning (always-on chain-of-thought)
LFM2.5-350M350M (dense)32KCompact instruct, structured output
LFM2.5-1.2B-JP-2026061.17B (dense)32KJapanese chat (latest)
LFM2.5-1.2B-JP1.17B (dense)32KJapanese chat (original)
LFM2.5-VL-1.6B1.2B LM + SigLIP2 400M32KVision-language (OCR, docs, multi-image)
LFM2.5-VL-450M350M LM + SigLIP2 86M32KCompact vision-language (captioning, object detection)
LFM2.5-1.2B-Base1.17B (dense)32KPre-trained base (completions only)
+ +The Deploy panel above covers the seven serving variants; **LFM2.5-1.2B-JP** (original — launch without `--tool-call-parser`) and **LFM2.5-1.2B-Base** (no chat template — use the completions endpoint, see [§3.5](#35-base-checkpoint)) launch the same way with the model path swapped. + +**License:** [LFM Open License v1.0](https://huggingface.co/LiquidAI/LFM2.5-8B-A1B/blob/main/LICENSE). + +**Resources:** [Liquid AI blog](https://www.liquid.ai/blog), [LFM docs](https://docs.liquid.ai/lfm/getting-started/welcome), [LFM2 Technical Report (arXiv:2511.23404)](https://arxiv.org/abs/2511.23404). + +## 2. Configuration Tips + +- **Reasoning parser**: LFM2.5 reasoning models wrap their chain-of-thought in `...` tags. The command generator passes `--reasoning-parser qwen3` for **8B-A1B** (it emits an explicit opening ``) and `--reasoning-parser qwen3-thinking` for **1.2B-Thinking** (always-on reasoning). This splits the thinking process into `reasoning_content`; without it the chain-of-thought stays inline in `content`. +- **Tool calling**: `--tool-call-parser lfm2` surfaces LFM2.5's Pythonic `<|tool_call_start|>[...]<|tool_call_end|>` calls as standard `message.tool_calls`. The original **1.2B-JP** does not expose tool calling; **Base** has no chat template (use completions). +- **Attention backend on Blackwell (B200/sm100)**: SGLang defaults to the `trtllm_mha` backend on sm100, which is fastest for the dense text models. The **8B-A1B** uses a mamba-style state cache that runs on a page-size-1 backend, so the generator picks `--attention-backend flashinfer` for it. The **VL** language model also uses that state cache and offers two backends: `--attention-backend flashinfer` (keeps prefix/radix caching — what the generator emits), or `--attention-backend trtllm_mha --disable-radix-cache` to run the language model on Blackwell `trtllm_mha` attention (`--disable-radix-cache` lifts the page-size-1 requirement, at the cost of prefix caching). Pair either with `--mm-attention-backend fa4` for the vision tower. +- **VL vision tower (`--mm-attention-backend`)**: on sm100 the `trtllm_mha` default is fastest for text but applies *causal* attention to image tokens. For the VL model, pass `--mm-attention-backend fa4` on B200/B300 (or `fa3` on H100/H200) to restore bidirectional image-token attention and full vision quality. +- **VL multimodal feature transport**: the generator launches the VL models with `SGLANG_USE_CUDA_IPC_TRANSPORT=1 SGLANG_USE_IPC_POOL_HANDLE_CACHE=1`. The first moves the processor→scheduler image-feature handoff onto CUDA IPC instead of serializing tensors between processes; the second ships the pool handle so the scheduler opens it once and caches it, instead of opening a per-item handle on every request. On the image serving workload (1 image @ 720p, measured on VL-1.6B on H100 and B200) this pair is worth roughly 30–50% higher image throughput and 30–40% lower image TTFT vs running without them (measured on VL-1.6B, H100 and B200); decode speed (TPOT) is unaffected. +- **VL-450M memory headroom (`--mem-fraction-static 0.8`)**: with the default memory fraction, the 450M's small weights make SGLang size its static KV/mamba pools to nearly the whole GPU, leaving no headroom for image-feature tensors — under sustained concurrent image load the scheduler can crash with a CUDA OOM in the radix-cache free path. The generator caps `--mem-fraction-static 0.8` for VL-450M; the pool is still far larger than this model ever needs. +- **Mamba scheduling**: LFM2.5 runs on the default `no_buffer` mamba scheduler strategy — no `--mamba-scheduler-strategy` flag is needed. The `extra_buffer` strategy (an overlap-scheduling throughput optimization available for some Gated-DeltaNet hybrids) does not apply to LFM2.5, whose convolution blocks use `mamba_chunk_size=1`. +- **Hardware requirements**: all LFM2.5 models run on a single GPU (TP=1) on either Hopper or Blackwell. The 1.2B / 350M dense models fit in a few GB; the 8B-A1B MoE needs roughly 16 GB for bf16 weights plus KV cache. Multi-GPU tensor parallelism is not required for any variant. + +**Recommended sampling parameters** — pass these explicitly on every request. Some LFM2.5 checkpoints do not ship sampling defaults in `generation_config.json`, so the server will not apply them for you. `top_k`, `min_p`, and `repetition_penalty` are not standard OpenAI `chat.completions` fields — pass them through **`extra_body`** and SGLang forwards them to its sampler. Do not set `max_tokens` unless you intend to cap output, as it can truncate a response (or a reasoning model's chain-of-thought) mid-stream. + + + + + + + + + + + + + + + + + + + + +
Modeltemperatureextra_body (sampler)
LFM2.5-8B-A1B0.2{`{"top_k": 80, "repetition_penalty": 1.05}`}
LFM2.5-1.2B-Instruct0.1{`{"top_k": 50, "repetition_penalty": 1.05}`}
LFM2.5-1.2B-Thinking0.05{`{"top_k": 50, "repetition_penalty": 1.05}`}
LFM2.5-350M0.1{`{"top_k": 50, "repetition_penalty": 1.05}`}
LFM2.5-1.2B-JP-2026060.1{`{"top_k": 50, "repetition_penalty": 1.05}`}
LFM2.5-1.2B-JP0.3{`{"min_p": 0.15, "repetition_penalty": 1.05}`}
LFM2.5-VL-1.6B (text)0.1{`{"min_p": 0.15, "repetition_penalty": 1.05}`}
LFM2.5-VL-450M (text)0.1{`{"min_p": 0.15, "repetition_penalty": 1.05}`}
LFM2.5-1.2B-Base0.3{`{"min_p": 0.15, "repetition_penalty": 1.05}`}
+ +## 3. Advanced Usage + +### 3.1 Basic Usage + +A single client with the recommended sampling presets applied per model (the examples in the following sections reuse this `chat` helper): + +```python Example +from openai import OpenAI + +client = OpenAI(base_url="http://localhost:30000/v1", api_key="EMPTY") + +# Non-OpenAI fields (top_k / min_p / repetition_penalty) ride in extra_body. +SAMPLING = { + "LiquidAI/LFM2.5-8B-A1B": dict(temperature=0.2, extra_body={"top_k": 80, "repetition_penalty": 1.05}), + "LiquidAI/LFM2.5-1.2B-Instruct": dict(temperature=0.1, extra_body={"top_k": 50, "repetition_penalty": 1.05}), + "LiquidAI/LFM2.5-1.2B-Thinking": dict(temperature=0.05, extra_body={"top_k": 50, "repetition_penalty": 1.05}), + "LiquidAI/LFM2.5-350M": dict(temperature=0.1, extra_body={"top_k": 50, "repetition_penalty": 1.05}), + "LiquidAI/LFM2.5-1.2B-JP-202606": dict(temperature=0.1, extra_body={"top_k": 50, "repetition_penalty": 1.05}), + "LiquidAI/LFM2.5-VL-1.6B": dict(temperature=0.1, extra_body={"min_p": 0.15, "repetition_penalty": 1.05}), + "LiquidAI/LFM2.5-VL-450M": dict(temperature=0.1, extra_body={"min_p": 0.15, "repetition_penalty": 1.05}), +} + +def chat(model, messages, **overrides): + cfg = SAMPLING[model] + body = cfg["extra_body"] | overrides.pop("extra_body", {}) + return client.chat.completions.create( + model=model, messages=messages, + temperature=cfg["temperature"], extra_body=body, **overrides, + ) + +resp = chat( + "LiquidAI/LFM2.5-1.2B-Instruct", + [{"role": "user", "content": "What is C. elegans? Answer in one sentence."}], +) +print(resp.choices[0].message.content) +``` + +### 3.2 Reasoning + +The 8B-A1B and 1.2B-Thinking checkpoints emit chain-of-thought as a built-in behavior. The Deploy panel launches them with the matching `--reasoning-parser`, which separates the thinking process into `reasoning_content`: + +```python Example +resp = chat( + "LiquidAI/LFM2.5-8B-A1B", + [{"role": "user", "content": "If a train travels 60 km/h for 2.5 hours, how far does it go?"}], +) +msg = resp.choices[0].message +print("Reasoning:", msg.reasoning_content) +print("Answer:", msg.content) +``` + +### 3.3 Tool Calling + +LFM2.5 writes Pythonic tool calls. With `--tool-call-parser lfm2` (already part of the launch command) they are surfaced as standard `message.tool_calls`: + +```python Example +resp = chat( + "LiquidAI/LFM2.5-1.2B-Instruct", + [{"role": "user", "content": "What's the weather in Paris?"}], + tools=[{ + "type": "function", + "function": { + "name": "get_weather", + "description": "Get current weather for a location", + "parameters": { + "type": "object", + "properties": {"location": {"type": "string"}}, + "required": ["location"], + }, + }, + }], +) +for call in resp.choices[0].message.tool_calls or []: + print(call.function.name, call.function.arguments) +``` + +Tool calling is supported on 8B-A1B, 1.2B-Thinking, 1.2B-Instruct, 350M, 1.2B-JP-202606, VL-1.6B, and VL-450M. For the **VL** models it is text-turn-only — do not combine an image and tools in the same turn. + +### 3.4 Vision Input + +The VL models (VL-1.6B and VL-450M) accept images via standard OpenAI multimodal content blocks. Base64 data URIs (`data:image/jpeg;base64,...`) work in place of a URL: + +```python Example +resp = chat( + "LiquidAI/LFM2.5-VL-1.6B", + [{ + "role": "user", + "content": [ + {"type": "image_url", "image_url": { + "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg"}}, + {"type": "text", "text": "What is in this image?"}, + ], + }], +) +print(resp.choices[0].message.content) +``` + +### 3.5 Base Checkpoint + +LFM2.5-1.2B-Base has no chat template — use the completions endpoint: + +```python Example +comp = client.completions.create( + model="LiquidAI/LFM2.5-1.2B-Base", + prompt="The capital of France is", + temperature=0.3, + extra_body={"min_p": 0.15, "repetition_penalty": 1.05}, +) +print(comp.choices[0].text) +``` diff --git a/docs_new/cookbook/autoregressive/intro.mdx b/docs_new/cookbook/autoregressive/intro.mdx index 4a9f7b41e..c4abd04a5 100644 --- a/docs_new/cookbook/autoregressive/intro.mdx +++ b/docs_new/cookbook/autoregressive/intro.mdx @@ -37,6 +37,12 @@ metatags: href="/cookbook/autoregressive/Google/Gemma4" img="/cards/logos/google.png" /> + " }, + CURL_HOST: { target: "curl", label: "Server host", default: "localhost" }, + CURL_PORT: { target: "curl", label: "Server port", default: "30000" }, + }, + + curl: `curl http://{{CURL_HOST}}:{{CURL_PORT}}/v1/chat/completions \\ +-H 'Content-Type: application/json' \\ +-d '{ "model": "{{MODEL_NAME}}", "messages": [{"role":"user","content":"Hello"}] }'`, + + // Reproduce commands for the Benchmark card's "⚡ Reproduce" modal. + benchmarkCommands: { + speed: +`python3 -m sglang.bench_serving \\ + --backend sglang \\ + --host {{CURL_HOST}} --port {{CURL_PORT}} \\ + --model {{MODEL_NAME}} \\ + --dataset-name {{DATASET}} \\ + --random-input-len {{ISL}} --random-output-len {{OSL}} \\ + --num-prompts {{NUM_PROMPTS}} --max-concurrency {{MAX_CONCURRENCY}}`, + accuracy: { + gsm8k_pct: +`# To install sgl-eval: pip install git+https://github.com/sgl-project/sgl-eval +sgl-eval run gsm8k \\ + --base-url http://{{CURL_HOST}}:{{CURL_PORT}}/v1 \\ + --model {{MODEL_NAME}} \\ + --num-threads 128`, + gpqa_pct: +`# To install sgl-eval: pip install git+https://github.com/sgl-project/sgl-eval +# GPQA's HF dataset (Idavidrein/gpqa) is gated — accept its terms with your HF account first. +sgl-eval run gpqa \\ + --base-url http://{{CURL_HOST}}:{{CURL_PORT}}/v1 \\ + --model {{MODEL_NAME}} \\ + --num-threads 128`, + mmlu_pct: +`# To install sgl-eval: pip install git+https://github.com/sgl-project/sgl-eval +sgl-eval run mmlu \\ + --base-url http://{{CURL_HOST}}:{{CURL_PORT}}/v1 \\ + --model {{MODEL_NAME}} \\ + --num-threads 128`, + aime25_pct: +`# To install sgl-eval: pip install git+https://github.com/sgl-project/sgl-eval +sgl-eval run aime25 \\ + --base-url http://{{CURL_HOST}}:{{CURL_PORT}}/v1 \\ + --model {{MODEL_NAME}} \\ + --num-threads 128`, + mmmu_pct: +`python3 -m sglang.test.run_eval --eval-name mmmu \\ + --host {{CURL_HOST}} --port {{CURL_PORT}} \\ + --model {{MODEL_NAME}} \\ + --num-examples 900 --num-threads 128 --max-tokens 2048 \\ + --temperature 0.1 --min-p 0.15`, + }, + numPromptsByConc: { 1: 10, 16: 32, 64: 128, 100: 1000, 256: 512 }, + }, + + // The eval set rendered in the benchmark card + "⚡ Reproduce" (the engine + // ships no default — every config declares its own). + accuracyLabels: [ + ["gpqa_pct", "GPQA Diamond", "%"], + ["aime25_pct", "AIME25", "%"], + ["gsm8k_pct", "GSM8K (1-shot)", "%"], + ["mmlu_pct", "MMLU", "%"], + ["mmmu_pct", "MMMU (val)", "%"], + ], + + // Per-variant accuracy applied to every cell. ALL values are MEASURED through + // SGLang (B200, dev-cu13) with the exact commands in the Reproduce modal: + // gsm8k / gpqa / aime25 / mmlu via sgl-eval (registry defaults — gpqa pass@1 + // avg-of-8, aime25 avg-of-16, gsm8k + mmlu single-shot), mmmu via + // sglang.test.run_eval (900 examples, card sampling). They agree with the + // LiquidAI model-card numbers within a few points where both exist; see the + // model cards for Liquid's own reported suite (IFEval / MATH500 / BFCL / ...). + defaultAccuracy: { + "8b-a1b": { mmlu_pct: 76.61, gsm8k_pct: 91.96, gpqa_pct: 52.27, aime25_pct: 45.21 }, + thinking: { mmlu_pct: 63.2, gsm8k_pct: 86.35, gpqa_pct: 39.08, aime25_pct: 27.08 }, + instruct: { mmlu_pct: 60.33, gsm8k_pct: 75.13, gpqa_pct: 34.41, aime25_pct: 9.58 }, + "350m": { mmlu_pct: 40.69, gsm8k_pct: 30.63, gpqa_pct: 28.35 }, + vl: { mmmu_pct: 39.12 }, + "vl-450m": { mmmu_pct: 30.56 }, + }, + + // LFM2.5 support (model classes + the `lfm2` tool-call parser) ships in the + // SGLang dev image; not yet in a tagged release. + dockerImages: { + h100: "lmsysorg/sglang:dev-cu13", + h200: "lmsysorg/sglang:dev-cu13", + b200: "lmsysorg/sglang:dev-cu13", + }, + + // Pre-selects the issue template's `model` dropdown on "Submit verified cell". + github: { + cookbookModel: "LiquidAI/lfm2.5", + }, + + playgroundFeatures: { + // TP override only: every variant fits on (and is verified at) TP=1; TP=2 is + // exposed for experimentation on the larger checkpoints. No Parsers axis — + // see the header note (parsers are variant-intrinsic and live in the cells). + attention: { + knobs: [ + { id: "tp", label: "TP", values: [null, 1, 2] }, + ], + }, + }, + + cells: [ + // ==================================================================== + // H100 (sm90) — default attention backend; parsers per variant + // ==================================================================== + { + match: { hw: "h100", variant: "8b-a1b", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--reasoning-parser qwen3", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h100", variant: "instruct", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h100", variant: "thinking", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--reasoning-parser qwen3-thinking", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h100", variant: "350m", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h100", variant: "jp", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h100", variant: "vl", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_CUDA_IPC_TRANSPORT=1", + "SGLANG_USE_IPC_POOL_HANDLE_CACHE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h100", variant: "vl-450m", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_CUDA_IPC_TRANSPORT=1", + "SGLANG_USE_IPC_POOL_HANDLE_CACHE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--mem-fraction-static 0.8", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + + // ==================================================================== + // H200 (sm90) — same hopper recipes as H100 + // ==================================================================== + { + match: { hw: "h200", variant: "8b-a1b", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--reasoning-parser qwen3", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "instruct", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "thinking", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--reasoning-parser qwen3-thinking", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "350m", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "jp", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "vl", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_CUDA_IPC_TRANSPORT=1", + "SGLANG_USE_IPC_POOL_HANDLE_CACHE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "vl-450m", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_CUDA_IPC_TRANSPORT=1", + "SGLANG_USE_IPC_POOL_HANDLE_CACHE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--tool-call-parser lfm2", + "--mem-fraction-static 0.8", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + + // ==================================================================== + // B200 (sm100) — explicit attention backend per variant: + // dense text → trtllm_mha; 8B-A1B + VL use a mamba-style conv state cache + // that needs a page-size-1 backend → flashinfer (VL adds fa4 vision tower) + // ==================================================================== + { + match: { hw: "b200", variant: "8b-a1b", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--attention-backend flashinfer", + "--reasoning-parser qwen3", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "instruct", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--attention-backend trtllm_mha", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "thinking", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--attention-backend trtllm_mha", + "--reasoning-parser qwen3-thinking", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "350m", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--attention-backend trtllm_mha", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "jp", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--attention-backend trtllm_mha", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "vl", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_CUDA_IPC_TRANSPORT=1", + "SGLANG_USE_IPC_POOL_HANDLE_CACHE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--attention-backend flashinfer", + "--mm-attention-backend fa4", + "--tool-call-parser lfm2", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "vl-450m", quant: "bf16", strategy: "default", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_CUDA_IPC_TRANSPORT=1", + "SGLANG_USE_IPC_POOL_HANDLE_CACHE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 1", + "--attention-backend flashinfer", + "--mm-attention-backend fa4", + "--tool-call-parser lfm2", + "--mem-fraction-static 0.8", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + ], +};