From dd2e4cdc9921043f15abff47006fed945eba3666 Mon Sep 17 00:00:00 2001 From: Yuhao Yang <47235274+yhyang201@users.noreply.github.com> Date: Thu, 16 Jul 2026 02:24:31 +0800 Subject: [PATCH] Add Inkling cookbook (#31360) --- docs_new/cards/logos/thinkingmachines.png | Bin 0 -> 26356 bytes .../ThinkingMachines/Inkling.mdx | 302 +++++ docs_new/cookbook/autoregressive/intro.mdx | 6 + docs_new/docs.json | 6 + .../thinkingmachines/inkling-benchmarks.jsx | 36 + .../configs/thinkingmachines/inkling.jsx | 1083 +++++++++++++++++ 6 files changed, 1433 insertions(+) create mode 100644 docs_new/cards/logos/thinkingmachines.png create mode 100644 docs_new/cookbook/autoregressive/ThinkingMachines/Inkling.mdx create mode 100644 docs_new/src/snippets/configs/thinkingmachines/inkling-benchmarks.jsx create mode 100644 docs_new/src/snippets/configs/thinkingmachines/inkling.jsx diff --git a/docs_new/cards/logos/thinkingmachines.png b/docs_new/cards/logos/thinkingmachines.png new file mode 100644 index 0000000000000000000000000000000000000000..c4c6771e9a285e647e6c1022601cc13297d9764c GIT binary patch literal 26356 zcmeFZ_dnPF|2?h}Dv^d1DYInnkq9LcO2Z~Hk}YJDj7r&iS16eo*(;TuLL_@9dv8AH z@qEAkgzx3@{pJ1o<@r*1JnoO%IM;K|^;{p%Zzx?oxc}sSA|j%L@^Y6|h=}&26A|q` zNVXe)l5b%+Nkn8+B7a%xwnOYx&*R7=cXoDm8ksu{$GMtzJ3W1W?|5?Rne$huE+=0* za^v#le_W@MuFYIKM(nNqV)6WyOG(ce6ZNH@Xi|C8j++acyLI@ktkef6utafm8U!!6 zY|fRHCv;{kJUSSnr0oug<{NW0~w5#pSF&-2~4exB@}d;NhCD$ zkZTgx1vo7<2{&0j3}IC?6N<6^y5y4IJeJ#OrYyOY7nP+XC-+0Z{Il0Sp_ryAd)?-h zdZPr0Zi=A$>+nfvZum8VVQ?gYtg-J2ivzjDMD z%%4509M5qh@lyYzQKvtzOt(~GnwEcFJsq|A_t-t86m0^W=bQOEo5mJ8rw0)Q;ZXPpjdw(1Y`3pv|0WzTMQ?2Y#!^2!zg#S*Il>32$T_|;PTl$!mi zLtJXv{n*=x^BO-SgJjzA`AepY73O_8dmDyJ?59o&J`5ck`%pn5Zu|F#PMMQB^R5Gq zTh%--?Wg`I#*3@hFH?64T~-q>xJoqW!TgR>^OIb5zs|89y{sDlQ?vpP0_!d@8GdR> zRw%IO(U3jxyG=yfm-6)XQipDgi_){1lU7+eWp5-XRueX97e-?&m7)cnJ7wpf;BY8f zB=DUZjL>z{yK(s_meYRnx32rvuf|AK_r()JW@_7;YeUW)U;0{pb>7PSaLt&RG*DM*~i>I ziJ$62*;t1>Uy?ffOMUmSBa?H8*z9~*WAco2eo&Q|%#PpupL5nMF%j3qOsW~4Ev5}LFWJv~f@6E5Zzm-s z#YX?~%getT8yf*W4Gd(_lm_INaCz64CdNvntA<$i$lcA-i3<-;k>*wF^`hm~zuZ3R zNw)EdDe2bqKgZVhr0m{@=zN<8=Wm#LDZlSJL_KdiZC~QZSo6y`;`-09`1nle>}(~n z?f$2f%afh=-4bYhekLnKeD*#->%-79_^`#UOf^fVVtd_`Zbf_5ScVq+yE5IA8|bm+ zeprX$ir&s@Kdwu?vrXI6JN?g=NzTN^Dj5ecUNKAyT&B@(KsMgio<_Bsr`5|J=C3}Lvtc=-{UVjrz9uujhk&5uyPQO z$A!8cE5iG-=yGcwKL3XMnKPzrr!P9?J)Y@peW(2H?U@pj=;i5>ne7?(9e)0ajEb|x zmVM=-k4A^Lu83+M=TN(W)4-}2{iUz=kjd25moyAR9x~S|=VhaD@P=Lqi`o4<`mK;H z;B<_TIiVgX`VY+Qyf%};Wi}J)qO~lga`z=y#Ys5!uGO)Dv%qVmuQ5j0Qtov5QoCk# z$*#%@|4*^J_hoyE?Gjv<)$2}7CTE9q?PO|~q$tHbbkh*X>gY;W&;4*GO(p4gOI>fi zX~(++w_>><#<|<-p)%rz^`X-}CBDUs-+t|L82s|Q^Jl`NvAD;qEG+pZt*rd~t(%`& zwiA9}I~lLMC|w=&%j2AAh7;VkO`7ssGkuvm zj>%jiiNE9Ie5$bqGxU3T#y2Ax5BM#jW1h|#&vzM zXluEPZI7zFGd-wPOEMPiZcFCOQ<7l*jM3xr*~D;ICU2uzx(K!e!0o;lmH5+D8QjYZ=a| z=NMos9az`KwnP1MzQ5fbjw}V&h6O=nGD(5 zSgFA0nkO^970`y`uRGIQ=1YEpaqP|~`<1n{@7!gJDjFIZnOa2-KM&$O?Rj>GFSn(j z&+Wp)wsQ;I9!!6AehEwgzfTaVz~Q*Iu6UREx)jBj^AemIKY~(jrCeWIY*kKv*PBA0 zx{H!tT<3b*aGF_c@)?0Zt-dY-2lXf8EUzM$dN|AL!DbCl{E-z+xLdGa?e zf5~lnNt@9|QHR<7Su-guh$Un(v@GSOzkjscnd@E4hb*1NV z>#bdqghgQGW#s42MM=kw{E0cTkmO%-zs4VXa?OHl{p`HHz8^K``Mc>fEf-ensh?sY zG#y{xO5G){vgq3Xgfz#QZZ`k@r)oWtjAI@SwdR_i%pCFJilairnRl1}{hHWZTQKVN z4Iy@ERTQ=xZ@H3MJZh&@!fVtR8F%4vbV8|+#j2{cUwy%Xmrn)gb9QlJ!S(SzIw7;) zK;RQcgA8-dRpK=yHzxTPjx?{ilw(~rQ?A?>u=+8VqC(tMTl=@Cq@XTJcyQmXtJNx~ zm+4#&t_4!Yidesq@ugtkobjyi&INM$)9`*Cwusms!HFxMuV+{3zwP24t#v?# zp><1=ym;=F$csl~Tdb9rgbsLo^^;xt_X`WW;aM)r^tJyLN>>!0w-lB788-WGFHfLi z2}(NgF)+LaEz3Flr2dM;+LYt`u*uIiSH8!JHZ@8nPON+E+EczZN^4`qj8FT0VyRe* z%kt9vlZFWHtF}Xjtal1j4iKq*%3NL@3J@-oZs#V~GHMktuk_dj3ROJysmp;>yzTfW z3VQM174GgK{9FwV?F5K%TKZQ?d6rANxUJZ30%f%}M9t&T9UuQ#(MPO;f_7{3BTDaX zv0f(|h-5gr?Wzf8;XG5;F>XsDQD>+9xTNRU`P=uHtzB?bv-6IeRkEG@Jr`vhzcw7A z>`8J!yJKg4BCD;f&HKP{S$qEbF-gDms(tF5jHLS7gA-%fRWnz5otp=&4)t&H@$>sq zoGO0s^9?E%yOor-?7b{7%Js!@%d(~L!Q{nfT!aenkUnkUx6iZUcA^de2I*$4p+74^7F$<1!<2PMj9d(#+rA& zUzn+VD%G#WKDJ!>WDn&ju`j{CK?aeBejbz#g<4mOPU=-jzIKp?r_$x?N7IhXtNwI) zMmIv7)mrnUo|0%6+txL1SmpeaN>xb~v;AB7C2gnpkw*T*J3x`dx0elWe0hH4?E3PQ z>3p)7m$%XG5{HTr!421qXkW?Jogg&5i!4s6L-u(5oA@+87>6&WM88W8x}U(;jB$9>=jb?Gk&zoe$Tr_e$~N zY$n_r1-#30mtz2O1J%Awp-))O;eniS)I@EDpS2l%qdyVQTk81e#8IkM+pvf0K}=Uw zR8&xYIwn|1_XJ51=ooXd-QOPr1HPx19mI`fHR;72tnmBu@}6HM9|vVl;3znuzFjX5 zWARPA`BrwMKJ5OP%Ri*3_U+q;lF@)G@T2@d)Ef)D59)@U2%UKDuK#^yS!NcNvW@8y z)G(=tdtsqMD4t*|(YnNHY|$wPPw)*lM)4bwC@b9f9>J|k%^IZ=D9+Oj97DxBAjYRA zd)c9^7py!^!Zo_}*fVkhve6#B`C8f=2RSiJPTtO!3S-rK&X33rIcgm+z6!tJa#jY*yH|r^R z%&nRkTgUd^ex~=LLG2+=?}W?0U%q_#WDlv4fk9*}HRM$o+pWnIXaBf`*-|C>iz-`!BAZQ&jm0_9$ugo2Q`b)S`A;w!5>6O3QIw7pjOl zj@`Fz*iE#X16&DrzrXzczIw=+CD3(1>M7g$50N23LA+S6oy{>xt_b$O`#6kyK%{Wr z4Mx2p$RlE7Hwj4aWpC&c8O94dY-}ctk-UNr+X_Js`i&{0*z|!h_}O9b`Pso4+cLNL z!+ON!autlbb-f-i5X+bx_Ho@I)KeI?(@pgDIc#L`0ROz{Bn`}l2h57*bPC=I-)+ps~a|i^BaFB*C21V z+d`28LaCQI*=_#fJ4JC-PO^S%Au2VwPKisWbQLuHySmdw1G8!%9~8pYLbD{PPV{AZ zp-+~55)jy6r7EsFU9FR+-8bh5CvRdQGR1@|A`PDaI?wo!3mxEorz5*wjqD3uHIEPZ zi5$m;(L~dDd5UcSKX`N-UPFZap_6$0cF}5D}K}m#Wqh$ktIV_D~)mkS)Dn!-@6`0uxXCsx4oJqofMORHl;^ zMfI_MUipf!{iIq*n#q~JTw(ne3~B>4Ez5olb99)tr$>o9@->^e4^_D32Dt1swj&h8 zXO2F4r3>FLTzPSn_Xvlrwk|{wwx-U0m)yS(I%Vra0h08_GFRV7-1_oLxK~8)Gx2x{ zS+R>xiV0| z{tV4UvXSh-^#X8xY8!aJz8Nx{OJJ&a_Br7w**4NEe~Tp^{797c-XEo|&^cg*1u{ZS zJ^J*?|Gd^=CWvIOeMf>rLYjzC^(9yPpW;B?94)2pe07BN`s!qEb7NNK%f#VOc8$Eh zwLwfx*WFoj&n-iM{Qh{KgNf9O3Y1hlFx0+7`!Qk1d0r-#H3rY3Zup;;{NCShI>n;h zq0Ob5AmM7TU+CKC0OODql!ne**@OHF?h1WZP@Jx3+8(R~T+i-2z9GBUwP}DeJ0EC= zvM1Fy5`51Z*AiCLM&`3|vmoyy&5e@mCDz#P>o*SR*!(j6&kUipC}enWQBA2MKw|yD zcuVr_ktbH5k~@}HJRUqBpEP7U%tS{~gGXK%X*e#RR{NjbqUux zc55u%&iK0onvj#r#srs!5-j3+pSO4j?$5~T+|bK`U7tOlC%zWg`LY|>61c+>Y^T0Y z4;pJNKyq7-+gMF5C>{S6Q2ZA*On-i;;rSZ@H6g*lIgiG$#gD%IE!ILIY?jzCLsjzp zF4QkM;kG$fLnttfl3Z((ZK=1)oR-!>1@Eultcrfh_$S}AV01sf0zJKiOCgjmD%P#F zORF2xIkN?g*-z4_JYNeGO@y+mT7ZuJ{NsZ=L*-d5v6sMS!r?#ReG^61I+j3Tr{f;? zo`44Ms->x~nx6~>>cA5V=hFVOc{*O)aULe*MXP>~sOUF(G!(w~zrHx;cbq#@HH*HL z#CK(OyjbF4+k2c$R9CSF6rZ}LP{H177Gy^8O3u|!*(+Ge<_Qz?1{@;a)D1a5qe~iB zB2hX_A08D|$4kbPN;y&&VvGy$D@`?``edRRE1SXQ>Ks0|i}(IxjR8lpDXMv3@6D8b zzJ7jw%>~{b4g5iXNYkd;j>Bs(oQrc`j^K}<{R}QL+~5M}OoL0UjJMVd4@`hy;$`NC z>)wvqw1aE0D1?*euUuF9gO9%j=b+(xh71ScIH(4E?3xPpU~(~ym>Hq*iU@5b$S{!T zmo&VLv*{`}rlq#yDlqE`%sPWnd;NT2s@5qRse}?|!AdAyY*By|+ONe?QB2~vIy(Rw z6>w7UJoV{g9B#$-sA!2hKirXFnd_)vG~JPgNW>nVa3gu$ zeT13FPF52x<;D|IdNK{p5XwnxH^^f!znQfn{W1M8I~BmuyUSfKa2t8A zSp7B=deKP7At zd-U;BB`L##uB!p`TO$#=(Wx)LiS1K{m~tju!jyAtQ7Cb?)7|R9UiC9Z-Vtr1>~}u2 zeY{_j*lcEV21Fl9*=26#QIGvBH#87E|}k*-W@<~9a18pgM1oKnbB#w2o4(2Ed$g)?pC zeh?(wyu7@BH}=^`1b}7%|LMrim_iH=h7%o4qDe2=r=<@nowFSU{r1Yoo=3Qq^&8yhQ*54doPxPD?8^lB3f z;5Loe;k9zVeMeX$JJaU1gi*{;hu1m^N$wqHpz%1I)!S#=s#xty8L`uP{bC_#0x2c^ zZ3Z>1i{{yNBSHpzuLzdEa)Lz8y6xppxA)nI#5+$H!HB3&oJy3*+W{wCp6c4@%k=C2 zw)-R6pqIk#&CNG(3!AijX?&od*t)$^v9mPY)0NWbbqg?oJB?e(Q*_8@I2V`)C>*2k zeM(2y%cM{94a@9zjNNF|Hot2MOmX|~+kpAeriLd(jf5)e-tyCu<{OTLKH5ZA1n%1{h1Her*f=r&4>C9juby8JWaE6AP zGEQBvul>_zEos3#AHZ^uRwz`8N~$jqog&}>prr=Nlc_N3k{O=)hIfhyeZV_ZQ9;vo z^5})6;^k^eNe<0V*S&+PR5+TW_{Z>>9}U-1*}f`ETWNvhemvVRWdwRN5ZJ?xw`1e_ z`{Z}FUbXwqhVHTWPLFI5gd5)EZi?5h(?Lt2II{>7ua>%8inF_$9KG@TegMHJ{A+9D z9wQ{e!t!-!?oh>NH<1_AoPQum`1jh8xGg7G@H z>CD&KH7yBvYvk$eNx5%rVT@qOdL0Y#wo+^*p`>?NUo;^h^^Mb`fwR{b!JV*Ht)~vv zcGJbTWo)7D#_3Bx*7w}->wbFV$dR+xBhOTM(zyuAf8;J-)qhHIuu(Z8Wnvwq2M)%u zB^6yQdVh$%EP~jLSj=T*t8!cJG`@M_6N9!`aVFtBTF-Jkv3mZiVINUuZp0ceBjU%m z_|1;-VZx(S9~AaNmuPXkwZOc~sGutj3JhUtAe}ia8NYxKn#b$a&P+BCf^M49P1hHZ?XH zxcg}9>knc(kGEnIrDH^__0dcKuQAYhD?na7wjfI9Ki>KdWn&(O#(s*$n*1Nrdu{}M zc}D1XoZx3w&D7$vzMLmqZ(}-{Z_%@kYAdMj8md1F!Y%7h$JR_k)v3brw|IKaVY+NZ zXT0QDkGz6`YYF;qrR)FF_%^QDkKJ^iMzbgic`^0Y<#J5wKS4$o0O89m9#bk!2E;N5Bkcta9GW3US2cs3atgX58C%=w9%wJ zUERpYXu0Y4nLhlxKIrL!UMGd!1#2_aM;LCne4&}-3wMkiy&j3heY~(rYy^0HPtO6MxM0yO0TR`_cghJUZf-kUD?d*?Vx%Ro zIFJy52S57ZSumIfbiv4*g4aMFFow!YF+@X|&74Au?e3scgxh#Ft^vGMH`k~MSLSccPS^LjH5Z_iR*pd(L`hQ;oukY&YB(+IESOi0 zGF+u&ADYJ$l$7Xk0Qj55k1VNEY&HjukB=`+bN~+YRx7%!>S8A z>HjHOGzqlIT^!*;IlB2dw}}(IS5mQr(cpWKmX(R|{nlpdg_+VvxO9^vsZ>Y4AW}&*yOWam0cCaQ+pD6rEz^I+u-yCHHq1k)t_J&dOiZ;T6Kp`XOy7vf zW+==saM%5!zsGIYUCJN+`p^`2K^PmaUR$LhL^l3YL00N^x|%gS26X7(9$P-o=z)BW zszx?%qy*T!g~*L(u{-7A9h*02?x+8x@1T76wght+vB z$`}VF6pZ1_wPA%!c_ zSjx#Pb-81SLB@w*Gm*cj^>1BDj?np_niYSgJ{-o(#8UhFoAOV^`%yYR2{ju+QQ-{!{E`C9)#4mIlk$njh-Z(P)r7j$xP{jf7@sVkI&VARS+_hKv-xh z zVbIo<3)=i*Za$V(I)4p}DkFFJHAk=t?po2KF@TBsf}YC5N^CsZF`|~A*&1xdScufn z2uDv3jW^kqzZDcO9>Bjx(&M@hZKp5H*$L1JP8y#oFUM1vlE@if8-Yi(^MKxrHu$F4H0F1SwkP34z+?uN5`ms5+% zoEVKTEs&!EpY8r%&ypMx#U*ChW*T(U8xD2S+OeCa=~E3!F?% zO+9?35w@O00=J;x5;|$>F}$1_7nqa2F`wUc+g}a_LGsM!0V}oP`rwt}hwbS_xY?ZSDR0xtG(DOezs7V?m*q+I zfSUV$XG(f3Xk^g^ltB9cVUVb`BWnxmn`IijIjT+0{#A z^1a2G;ORLoYx8)NCz9NfV{j6wG_PXcPXep}aivxRo}7Qq-+rHse}rY_-9t`RMb?tt z3u_z3a>#k5A7{5oPrLB^i?g1eg2RQo>GC60-Y~ZzlZ%X+2JN26>y(rsy-5&Ud zH_kggsrWNsRj8Gz|Mi9UME}F+yC`Kam`HWx8Y^z7-&Q)GdRHv&JMaB((D7Gns`EeH z70O3%1MEhx^Us$rU#^9n-`jI#UKcA7Cv4eEHp5cQ1EB^CfjBbD$zhLKfd?3e&Xd)^ zXCqeHcLu zf4jcHC3W`2{m2mlIEPsU_A`{QBdg@2zm0x&DMYkUHq;oiqd=^ zJvn1jwek+#TLQVPV7N-}_~)w>$_Y+eUS+)aW6`+pW`7tMUtaj)a-UK8vlHry%B{vl^F`gf7m+4w({M6 z0$jX#oj1GN2$!Jb+zZ{f3{Be8SI;u5|5a-I)Bf$t_{pV5zXml;y}ui?(=n+L|4-0T z?>Zrn`F|PA|DVIC{!D*pJ$7n%$QL%GTk|4+TU*b*ZQLO?wQ&DUdCUIU?@YhfNBCaE zoXp7bfBp7IM#ii4WT&m1jmj&g$FFR8U2cD1Z3SP8y!zRaJp+ZC^bJCO>}@xbfwwu) zp@}!YgWp@_K2z|zmu6L@Wh61Wuy**!k^3V>tyA_JF6D_a=Q&lSr9C-?0{q0~>#d)2 z=8XM!g#Fu3AS!{g$FRXuQcyLqU_W2#$5%r}4?0!0G>dwZ`*v347+R=%|Ws)wL5#s6tfU z_o*0upC`1z7y04iPT5NW?aHHD?@u$D|2e_DgORmEXVztnE$p6#@Bi&bI z_xNO~$1uq4(J*EwECRww-@}A)YanxoZc92}OPN7Xk0~8=n2%(nICezL#8^Ql%9EjFDTnS?% zBiYCIQJzMYoJwrR2j?6n?VW|DE*5l^nY5;0<)Svhk;j-Mzx+t(_%A=;ek`2nT*V^v zB78c&wX=HvL&djww*Mc`49mP*;P_HT0e>5wPInUDXqv2EzTN-@D-^WV}5W1|21^!>r*Au3v_`_QnHms6hxk>Sy|c?=PkDuO_4zj?@-`F=4w0~Yoy&BzJ93A_^42}}h%A00Q0*bj^Q=lBdPv0&%_}oxmw|HfGj4j~Q1tIIfFNxLWOv=$o zaMZZL7VXb21|9B4$dR!?lfOb*A{TTQ!4i$4fsi>8&|k~m(jw-2?6J}|1wZV{Z~)PO z4Pyx4b(%k;-_-y9=G*f1f9$w*y&V2(3{$xkBpj~Se*U3xUcMq`G$Zfi@Kl zgUGVnw`iF32qbVE*55V))YJr=4(KTm7Eru^2ZLAEsQULSvQf-3q8W~=;a5WeBtn1+ zQ1WmwN9Lt}#qq(xd=Sw#Ji8M(g<_^a>=0!Rp5vV4`c>Ftky+1KRPi27Ts!MB%-1jFAveB2`JJ^x|NC_)MO@YNRAH=Uxop!&Kc+X5SfA3-^-cv z8<8T=$llL2R)-vqj@#M@&YFh#!nP7?}v9R^POLbzivv9-0?oHBjPyEJQW#6<#|7NI0Kgxj(xALWbW>On%oCJ8S8jI z(J;U2zy+Sg&G#x+O=g-LOK=Z=s-LU&%K#UTyJB-53(uhw75J!YY{BkdoBY43o9FhS(ecZsphP|NpykA|?Znh^6 zqJ@xR^?T_XB9VyxjahGLr&!#r*TGFXpUgUu$jX^rvwb*X^)YEn%9DToetcIkL31&lFKs$VX(Hjc#1F*ZtfvzTP*XsH|++ zUHGxvu(81C=Q@K5?Voa2XZR&(kX5glNO|?EvBb^5PeB$hMJxQ~OR|>ZYh9KV8S1&L z|DKCC7?8R_o1%s^YRrU@o4}mggb04yt9;h(Ss1&jkog~h1G%r|!&pkCqqjPADV5~q zqj>bBZB-XNzLH)+v+XpYP8*;7enhg`DaR!g)WUP`0Ry4DK#g3phxO9+(}esQ`q~^9 zUWe4HaWMJR#V(_vZe8KF^|{2&gc>`9+ebKZrGDq%W9RD+d8Tg_C5_?#B9?vA{>trd zX>tuEKn9ilU&{^P5q94)LMa`upV*H&> zUEm4-W$GOTIl1iHrm^{kip;hEyaKq*ExRMlnuY}eiCiyIZCdW6q$e7<@h8zUEcIi9 zkDt5Izb?FIK$^^7uSI|kt`A`Jy1e}JyS>4q{Q$tzF0)TK!^ex`kqC^Bce_}py!;LQ z8T;w(HTcU{441S`7L!S0y@@$y-dtR2y9Gm(Of^0*Vu8?`M2*%icj;`r*7jM~cD%bN zH%X6?VtYxPm;5cZ7KTp=f>FnG^X_J5AxA~%Rf~1_>Cq%#3kwc!-_)&p4PkyC23+#= z{T#uC;~vh7f9F78bd*ho9?n~$jaPzeKv7)0DmhVw1AkpX;lPa3P$9inzr!p+xgn-U zX!~8G2$y{brH!~g-)W0<+5O}M2%0MB%XD{GM3}5SOLvFxVMeSTM1O+dA~|@nHpU7T zpGiAx0?a`S1N?j+cM+10$oa~No^5l=PB9?_!k|q73P%G(3frIdVW)or(6Q!#FF&t} z2So@1!)7(ejSuwq%R7wjiJ@HcLQ5%)4=?f}=u3Bf&H!P6x7Jr^{dSFxBcmOu#pYO1 z8M03l7R(jOy3j}aIT0{^UN_{xzCz82J*K_ zd}|*(fh8zkCD?W{yJpDb*N0!er`;kiJhlLpJ;?q)08nVEwQ?!qODrOv*4`Pe&Q?=`ud*{$GOF(c-X)|O!Em1UFzl$vis7ik zM|&#k*jctqf@xW&CpK)b67~Y!p~he@2;l;N9bFp(kJs^ zC(w!6)?$W->wS!4m2`9u0Pd=3yZ?>8-E$otSR%Gg;S1krYqdKkmabha1dC=kXhznN z#`#Qj`>O@wj7$fpFFX4kx~W3+Jw|vY&!qLwCv)~+uO*YT z@--uI&7PvoB#+~D3O6%1p#WyOi}+LayjWD`RLibxEHwh_6=USGRH^`Jg^Rtp9>GC&Xy&o)RuU)f&? zI{^TOK6&ucX(=jue3Vf2#+(wikP;otAQ0@&3|+5D88ZKb!5V^rcRljZN8g(#OpzO8 zP(Q3ZhyKXS>q^Q&oJSkss#7u9NDHdOvwy0>)(x!}9|7pFMa|fCjCn z&Dm#hl6=H#0B=jo7Rhnf?Rrx8Yh`PtkCIl1BPE8_;8o4g^WSEJ(J4he7W8@g#~DPF zEd&tl5%Q2I`tt~AG25}W}UZ59axxfW0 z4=w!p=n76D`Pd!^6p+!z>wh(250*METssEA-jRCC7=W?MA~Pqp7t0gDp{@eeM7^=4 z3QZGi7rA@JWzFMdw@1#fSQ#2HXqpy7idUWvXJ3b-3Xc^vBslC$YBfVCQeGeRy&;M?xx&%qdm45ePlsYiv8C$VEl8SZJ-!86nAITVK$lhJl z9vmEuYjOj%4FsEnKAPLSA9r00zm8ka!zP$B6q|l0jrxL+ES2Zb3)>yXvfdN?Mz{$w zvJ*bSymGTWnAZr^(|?>4w;2eP1zTOa#j1bx4)P1=?q9V}XZC>a$Mg3QkzI1uPJ>*6A$&PgO z1u(e1^xWqP8ZiSQidG;WwY0PqZUgc+^85;SoWA3(m#MNM-9!Vu7t6u{64P)mmQQHNxj@A|A+@Uw3`R;lj9Yj6-_0ZQ zZ^0k#;NYen)Sb``q+%mXnji{JUfutz-@QeeDxRN+S1I!g0x^UEhU=&Ow0yu}ci`pJ zRuK^so<#W?T4tudJ{(KziFoo3Hw}?RLmI!T!@CG77GUAd!dmGI>Sc}59Zh~>KR~Qb zsRQq zNws1Hi|4c|g1QqL8d~|P@JmkWt#|PlbFe7zO)iRX=f8MSyM0%Adj*@{ zrwFh+v^&G&eq8(f7rMyPO~uU!d4O$Emh7XfXqmWSUWO|k#$8zxGllMDo9lW$k_{($Mo08hdlllZy`G?UP#(7y6)N z5kq5M7}K2yv5867F8M_6vE9`?7qBm=^Y8A8IHtbf@bGVRRF`Pv9}z;jfM3!$&l{sx zH_%T;&L#5HjKZ1Xq%iP$3?@`e*@{(fKNK0HBnWx#O$F|<+kkGOM>^{fGMW``t{%_L zngaRWbxjeb5Sb5>t|m1Z5yo~2Q-p;XP5K7M3Q&~HdrL*FPN=mCAjF8V<@0wwXyNH< zG7tl_v9alIziUxIXz)S-f94>)R9G8gjM#cFasnWy2=v%$Cx}=N0zpPz$K9tA-t39T zcK?GP>UEFa?vTd_-+s?v1ILMsf=>^Bov*w>DxZnM0JIHxV-}w9e1wv4kDFqICyL!N zo>wL31WUSahlUkF3=Jd+`u9KZ*|662@Gh zc~`i1llQOJd28W_z~Lj+J|LrFBP%P5LjX@x;Uf>L*PkaTCV%>?o+5;9CwdyCMVK+? z(djdqhdl3xfrEE_nnIkKE^(gW)FQ30Md;F*96C>yUd2crhFKb2lO%A{TX_#|HYn|9 zs4#?QDYfFWn0_Zi;}^zlPW?Oc=rZxv!w+FTK9ELkHS|BAQ4<-inMMr@FMbG~df@{Nv~ z`}rN6EMcG$ObdFRdzJa;qgBWyKpsNbZppr+3CiaVK0A)&r_!~^UDRP{3WJWpEs3Zu zzx#;BBpUj!2&8ZCsRN+?Xib4P?}&1Lk(9*pji5h<_8oKO;G>+gD>HpNXlmWb)ao3X z%XSJ38)=NXRlyjndtTSrh|x-`sp8oXB28##Pumpjj&MjehX{3nL5u@))Yd0oEi^Fj z5UJ!f3l@T0L@o<%_d^?EKRPEgT*18It%AC8F)=Z1WEzTYU5nu2?8yz!*fo$qR`u~( zW*&3~Z0>bj&yI(4EP;v%5?dJsy>Q~ys0ULo7a)5p!q3ltX%DkhDO%<$-H$M^l5bQR zqS;d)#*W`Sxk}HrlaEgXE<-`*IT8XyPe+a$xt!fz_ljnKgw`Nyd@hXFB@gJ0>JD2X zMAhjuv)PelFBXTiWY_^t9ly{{DUEkEr`=1jW_Jh21&zmv^vR7K~a8otF{E_?F^Nz)lxfcWsV*KpW=R>3V zq03`k>~G3AKXqA@WWY)4SP(gDZU#mO2DykNC;bKc#!cE+e0p-k@%IH4;a~$WP+H(H z*$jSdI&rLq<_KNdrm89)o}fT8YFwYI^X*?oR7RaPFAw?-k;c7+Z);U0v6(1U6yohMa6uFY<}_~3~ETC2=|Xa8Og}U5!%G~7SFBI!EppsMT8-GExXa6&UBpG z`0|}Rch~gbv%in)I1J#<3G&~9k%>Ek9NMc21j_BR-yi7ULY@*aMBGz?F@>}(r^ri9 z`@-)1gaPQ0e-2`A2M;rde3DP=fkLiSjowsMXZh_?%|MxhXIzCPx;wp`$?gYeM;NDO zd;O_Pd~$@rvnOZkR(3kTx6U#%M@}+_ksfwr)$jJ*2jfUHmITmt$$uYw8C8iKOM{h?> z6@suaT;zWceHUFN0uP-H@wokN8~z{U{c1<-oCy(vvIgIP);t&%h-zqZ3C;Wgmm{bSHE5z1Qxc0^4DKVZ()r^$0;AC($Boyes)E7PjU*!a~{oe<(v3mIMAGY;9sU|gp z2z{-WE@8E1Pt?`mM90Er|8eHr+6=fp0c<=5j;N57V<;s?z(jwfvz$4YP=!$K6#Geu zo+p8qVJv`hX{>gZFiaMKu{n(f4&ALQToI;t)lgE8$uk+PKzuP6Zi(7rE#6a0PH5^l z4}RHaW0n5WuMo=WIJXYOsr-TE5Aqc40-Ts7cgi+|T6&T<19axn$a~-5#K#?_#gLtmiisYPNEqq z9ZpV8#cOv*EB0QJi!eYY55b!gw%Ot0h4f;!=kTV6*YuHH9AZR&mdkYXUQx#CMAhPk z`oYJ2t0zvx#>RF7M~MV6iil+j-5a<`9Fgo93&BI8=>nKF62fa4` zRSsP;NVuJ)qwN#teR5)C0oUc=dUO$;A7P-QfxAK=Dsq>>fRn9Q-2HDa<&~6#Zc!p= zh+Y;PGNB5(r(q`Tfjvw8_*Cd#&4L+5%vCKv<54f2t?=bs>R%Pf8r(Ks?iKiTA zQyG4KHa?I~=6xSa)8MJc-E(kb_j1jxN8CUX37zw3TkU|x9)X4x)kx825(zVzCDSJn zc>Ym-%7ndC#}Ybz=pKoT$@Clc-GX+hq>m=%n}Ts3NtoV2yvc`Z+nV)e&bH( zsWh27k1{=xC)XNbIuzehVNhhOxSfsJQQW6}>tuSlo$Z<0U2sY)AI)61idO~fhRFZ|Jx07n#!`^-VBig zB^2Jp$`wc}4`NIR(~m5&7UZEm!sZnG%PDp3I4XM4w(f0IaqvrY#<(s`hDXJe$ByC(ULDnyPm!q0Z0z{>U_YXPleZWpw~W zQYigyM)cmR(IhN&lbyy1uAu2}N>0GSD&N`Oc=}zyeT$C5w|T4tKP9^y*+sa`xUB9Q zQ^Fxkbi<$Cm}r1p@2hH;BJ)4pk^|X3h1ZM)PUdG`Upp;q{wEJZi%~WGmQR|;7UVhC zG|HZA3s*vCjwWc=8VwvOMMvCm=)!F2A-*p~=%Oza72%hkUoFj54b*=RD zf?&;~bs;tN&1*Z?K7G2Ks)9^^p1KUNeIi|Z?aM+zjn8suBl+Uo9I5O%tBY#bi zRwif+$yWB}3m`Ac*|WOM7nRp_rtF>Z+Mf2Q4fHy&XP})QoXC*v&9)Yc^Ct_=LLmC* z_R^fJQ(3Y7)Weit!3wLWh9So)^z)cuz!{<;U|(|a=CQi6cP1KNGH(UcU>FsE+urRo zk-=bhTum=>|0c&t-9SM;vZbJ_^o|JD@8OQc>Bbk>Y=H!A^S_6_;s)`q+$7=q#KywH z!pr-!Uhr_|Hu~K$x`~7c>VM|(grlCc#Ty_gg@#PVQp)8@vjAoG@N7^X$wCiUZkZVayzR{v*IxdqA$h>~8t&^i}Z- zcUVIX=b=Sh`QZ>>n^X$xtg1`$&mDNZpw2Y?mdWH=h+5-8)E!mGD$pI+8#j4fEY|`q z3Dy$91f=pN9P@S0KV!KTN|;dl;hZs9iREtb(0E0C2P0$&NLhdtY%jE`uR(`@hK)4cXOD^?|_GR z{;T$Xi3jwQ>WJPx8QGco`PR2Bjf2)Rx%Qt;5gdwpqt%F5@bX%N(BOX^r=^DH`oSHf zE>sNb4do&>3(=llG);9#dGGFO9;cP*EpR6MgR`?YNKN{X5lF*m9-(6LUc5z|^L&RT z5L9~#(}CCb&^;onvX*;6$HvA6-EHI1jEN}1G!gh4npr_RbFiWatk9#IRdp3su@=-Y zS?R3v#bHtl=vV|5r9$3!lff6?K%hR`QvSyseoWrX4b|ZLPH@`mvfVFuRFt*F0O|0D zs%du>*PrcHCQhALM@SgoMdW~xd(B`XnSrokD<2F6(V?M>=vkA- z>5n6g5HD;QpE-fdjccIbrEN$rDbH4g@DCk#ga1s4xbvfKh6?G8C14YQi=BPIC9RJ|E!j1|WCrPv0&c zLQe``A%uDkaAHxI(VFPX>okLghAkcucGi2=xDx{}()+vGXTF3U=YR`RBW-xepq^N` z+o~Fav|9=^ggGBVI^h4ccjf<3=YPBsDdSwpeILoufd*}C>j)X7Lav-sk|RZ|MWS+* zJDDWOXyrbKqIE~66h%T&)LPoqP!!|){PfuW;rq+}R6opPe2(|~^?tsdujlhQaAZg%M^(Ii~)fQM)R#PUe0`#)cC@(%-V6#+yv;=IHw(ycw|ru5OCt2}NGKZI zGV*P2V-v==;6B&eT*x@M{7|i2Sk+eXuV3!|l_oO}nS(zpvUdLVK@x;j8e4CQnDHiD~lZ{zY-xu(s^zex+{8X{9?kc1|U zl#C}V3+Xm($~0W{KygUAG3UEnFcT`q;g+DS)QB-LYc`^yDq+Q)jHLL$qao&CJw;FlmnCvpxjAq?ZWTfC%ScSJ-C>Lo`(|EO$-Z$k7i@5J;*( zv=4Y3_ya-Pikw~~q0aQ?x`1T{Uz&?pi@Tm7ta6PgcreqePYFXcfNe3%Af1dBkX#|P znNcVx0X_plH`Q;BfV2eXDMB`4w8N>BW1tn5o03a(NjZkcCmW1iJfU+6Bp zOQ)K3!p&8{4_zKBlw+H#?aw!kM+{(%|ET^ka+bTOYxxZKwT8{Ii^;xLu0>?i2KwX>9QL&&%mq}V@fG9;E}SUqJQdlEL`vFUTRq? z-d?4H)5}l+F#l}yFpkY$R=zO;n<<|A&Bv0+6zuh->GyKQ1tt&qmN(fUorM*!9JIwW z4e5Oi&M2s5zCBE4MsK^#I@$coxOFe-8m#HIsnZu|9+Q!?_+o6J=Wlqt7K|paK_eH& z*RcZ3ptOuV0ht@cKdEdQ;1w<=Z^e&Q2?c&*js*>(TGS-c14gLPZP^AtP6JJLuS0_^ zN3pwZ?3UomC_sc|0HkAvqmNl)HU#+x*n){u zT-J}j3M1JIs9VNL(`=idaKp>N#L27nJ@?Ehme}r@&9__hVROWxj-euGbvU0)Py}v} zNM;rm$4BqZW5+Qc@R^?=205-xcjt>0*z`1vQ#?OgliEV@Jh9_^dLjYCfkwigluQv$ zg<+^i2>~Gx#SC9d)owuIMAGfV+2K&3ZyIbbPLnRyQ7l9P1E6*6Y_~|S;MVW=$fA4l zwwhd6L-rq8Ky$&P^B9kAr?T?JtW^xmn^7s(Hbr+oGec901FU3lI=I34)d_eb8eKoa zX7s}p^!sj2EiG2MwLzt~VfOqKjvqB_@r9WgZ9;irg9Iqx*s?X|B|JpC556hRKL$`1l@8NDFm>tWm6BKl-73$%r*DzThDIVvI-AtPw) z^uoWCw2gQD!&b^ZPoMyR#%~CXoO+sI;yivQW%VjFj-hbg#!&gpnV=Uj9I81yd!VYGf#^}& zeelX0M8}K-FM-OJTCIr!B?3PEuaOOi2Bf*er6Ko#tt||G5^6Czylq!;Y-jJf!n`|+ z1hm&7!+xm^-R2SUc*0$9EJWL{uJWA^9mYqV?%M`53Jy3#5vCg71>!G9~xyJcsQ zle~o6A;!U=b_HXgi!>yR_A;5IVEa{asS|*5bij@W?Pe9eshU#zx`Ql9MXG?;X)zeW zZ}kg!5f%4r2V}V4p(#dsl|g}>en{^h_3bS-aWx8z(GVv9Sq|p}Th2z1BN%DBMb?dO zPHW);dp}xz)cOhUm#+>0#FRU4Q;X)hUitSIUdnzGTT=nrsSQ_UB$Xp;xKcal3TYR2 zt)cyK34-!IoR5@FX*-yx(J7&e>;9G3;ma`CUQf35*^*Sdd<=`ge_GG*B>*Q4o!odN zS)E*m_w_FM+z&q?8k$nj1xbnA-}%8AgHiyU`wB!QzBhYy=()ptOFI7^9pTCj zPX!nSd(0kuZ$Fh-eyDhWLLT*^&Q+NFIty?DwZJBA$L3(5&>_BxJF3s-PtBRsgc) zY;?(T6uVL%+6dJ({6j5DK%dd28RaZXcwvY(AL53q$+}m9007YiI~Jq|vA#Bl z13g5GY6Xa_8MM~S2RTSQ2aE!=ZUmS7lDiGmcumL$$LYp_8P;Ma>NS>h%5gQ@A~78J z1j<9N*;8XeeW?_7mN0ulVw_gH$9quTC9W^`;z5otzcnERjD{%6`nt7)$pR&kQ|aP} z9)+f3CHVI{?xN`b@1oLp9GmO2KN8^`wEbtXt^1)_>^iDC^c-^+L~nh?Sf69lu}MQ0 z3_PHX!6k9mDsC-3g}(=ZKX!RR(@vS)bW_c2og}Y|*ItvNc!|L&n1{C}O2%oY@662% zodVke-2iv+D(h!1^{rGT#54Iu{S~N-X33*^2wissi(k=}lT1>Wp&T~jXR*Zt!73VH zCm^k&B7jttgo|GHRe0BL!YhLO{l zoAv=|*f8w+wHLoG&l9F%TfWDaU7y(`ZfA?nq9D@O2&-9p$8iKT0TdB2PM}>G9pXGY z#|{rG`Y@pU-RVc0a0@%xVHutlL9R*}a77FgK9sfAau-!~WWVsnxej-poq2=D$FMdC zmFTr@2X-b4)mVN-f?nap7(Nm@&Z+fr!d&4Vv<0x_Uet~ounr2uv%{UspP5UEtP=FU z>EDQW&8PVtxG3+3)2BPm)ncw)~Z?PA6rw=v8FqfFcQW;(^Fo;A#W{-(*cMIO>RS0KKw{^6)%qO1%MSCHM<8u5XQ|BEd&-OS$#z+v)dgX)( zsg3tWf6Nq-L`;=rgjSlqtb}F~r=E(cYJ3RuTz$E$-J%2`DYr;f4lK#wC$=gk9(fb{ zoZ8KI6AeU_)!ru}&K9G%lJt840~V^b3CGX@MA<}le3Yzeft~XqZgTQu#~X>H`mJ0< zX$NtY$&*sO-LXIi^GLwz-(V!rFbKb1;|aPI$#jMNQw37h%G0^OoXr+kLMab;QPINW z)@X=EVLbKJFS$Ui6iOT_Eg+2l8=}ZyC2gMny@*tXhkpLgisM5P4H;!vQ}ZqBWZG?} z#7|mEOI^cZ48|xwX`|~*h^{0AGT3i}^dF(zAarNLUJy_Q3u7EUp?dD>iE3h%7M+HT=)|csLmwr= zdV(3}UWUrFo^qaDxVpoFyY6JD>ZwQugxQ_BJgL~hT94sU{mxQy_6Z(Q* zDlbQD&Xw|}FE7#bnD}k|due0zzd;4k$qzQnyG8L{J=t7&ZkXhe#^k9)q>c{~G=<>D z4?q+qT6^Y(o-uRY?|`58LGUs&=s!+t=X)W34i+}-dOc7pCusb-zE2y0X7IjlbCDA5 zWq?dFK_LEdL^ox*+~*~>7~$}+u@~O3t7tW%=mIdG3A!+=7!VQb)-IhrJPz0n#T7Z7 zhpu#I|E}3aHM3P?#JGC!(V2ulQgQoAQ-Nw5`pOk za3-Jr+wvNU)om6G+f|2|&|nCMe)5vr^TXYPTNN2_;hEjwh}{EH`#tP8nKAUl1}tt& z1*sHWhZ{MSkd8*pJ>9GE*|@a6_&voJLW6^^lu}-Gp*3e?{wTZp!Id#G@#7P+k7tE> u@b7;=S@GY`aQydsB>(3hX1R8pZ743;SuyNkZHGO%%G|`txa?QAbN>T_b6_3- literal 0 HcmV?d00001 diff --git a/docs_new/cookbook/autoregressive/ThinkingMachines/Inkling.mdx b/docs_new/cookbook/autoregressive/ThinkingMachines/Inkling.mdx new file mode 100644 index 000000000..4a62e0f0e --- /dev/null +++ b/docs_new/cookbook/autoregressive/ThinkingMachines/Inkling.mdx @@ -0,0 +1,302 @@ +--- +title: Inkling +description: "Deploy Inkling with SGLang — verified launch commands, tuning, and multimodal / reasoning / tool-calling usage for Thinking Machines' 975B Mixture-of-Experts model with 1M-token context." +tag: NEW +--- + +## Deployment + + + + + +For all install methods and hardware platforms, see the [official SGLang installation guide](../../../docs/get-started/install). + + + + + +Inkling support isn't in a `pip` release yet — install from the `inkling-support` branch: + +```bash Command +pip install --upgrade pip +pip install "sglang[all] @ git+https://github.com/sgl-project/sglang.git@inkling-support" +``` + +Then run the **Python** output of the command panel below. + + + + + +The Inkling images are being published to [`lmsysorg/sglang`](https://hub.docker.com/r/lmsysorg/sglang/tags) — watch the tag list for status. + +There are two multi-arch (amd64 / arm64) CUDA builds plus a ROCm build; pick the CUDA build by your CUDA version, not your GPU: + +```bash Command +docker pull lmsysorg/sglang:inkling-cu13 # CUDA 13 +docker pull lmsysorg/sglang:inkling-cu12 # CUDA 12 +docker pull lmsysorg/sglang:inkling-rocm700-mi35x # AMD MI350X / MI355X +``` + +For how to launch the image, see [Install → Method 3: Using Docker](../../../docs/get-started/install#method-3-using-docker). Substitute the inner `sglang serve ...` with what the command generator below produces. + + + + + + + +Pick your hardware to generate the launch command. Each platform ships a **Balanced** recipe plus an **MTP** (speculative decoding) tier and a **Long Context (MXFP8 KV)** tier where validated; the **LoRA** variant serves adapters on top of the frozen base model. Set `MAX_LORAS` to the number of distinct adapters you serve (1 is fastest for single-adapter serving). + +import { Deployment } from "/src/snippets/_deployment.jsx"; +import { config } from "/src/snippets/configs/thinkingmachines/inkling.jsx"; +import { benchmarks } from "/src/snippets/configs/thinkingmachines/inkling-benchmarks.jsx"; + + + +
+

Panel controls (top of the command box):

+
    +
  • ⧉ Copy — copies the current command to your clipboard.
  • +
  • $ cURL — a sample request against localhost:30000 to confirm the server is up.
  • +
  • ⚙ Env — edits the placeholders (HOST_IP, PORT, NODE_RANK, NODE0_IP) the command and cURL share.
  • +
  • Verified / Not Verified badge — green when the (hw, variant, quant, strategy, nodes) combo has been run end-to-end on real hardware; yellow when auto-derived from a neighbor and not yet re-checked.
  • +
+
+ +## Playground + +The Playground is where you experiment with **SGLang features beyond the verified matrix**. The Deploy panel above only emits combinations that have been signed off; the Playground lets you turn on additional knobs on top of whichever cell the Deploy panel is currently showing. The base is read live from your Deploy selection — only your overrides change. + +Lines highlighted **green** are added by your overrides; lines with **red strikethrough** were in the verified base but stripped by an override. Any change flips the badge to **Not Verified** until the new configuration is run end-to-end. + +import { Playground } from "/src/snippets/_playground.jsx"; + + + +## 1. Model Introduction + +**Inkling** is a Mixture-of-Experts model from Thinking Machines — **975B** total parameters, **41B** active per token, with a **1M-token** context window and **open weights** (BF16 and NVFP4 checkpoints below). It handles text, image, and audio inputs natively, and exposes a **variable reasoning-effort** control to trade latency and cost against answer quality. This page covers serving Inkling on SGLang, including its **MTP** speculative-decoding path and long-context prefix caching (unified radix cache + HiCache). + +**Resources:** HuggingFace — [Inkling](https://huggingface.co/thinkingmachines/Inkling) (BF16) · [Inkling-NVFP4](https://huggingface.co/thinkingmachines/Inkling-NVFP4). + +## 2. Configuration Tips + +**Multimodal.** The recipes pass `--enable-multimodal` so the server accepts image and audio inputs alongside text — drop it for text-only serving. + +**Memory pool ratios.** `--swa-full-tokens-ratio` and `--mamba-full-memory-ratio` (both default `0.1`) size the SWA and Mamba/sconv state pools; tune them to your workload's usage. + +**MTP needs `--enable-multi-layer-eagle`.** The MTP recipe drives Inkling's multi-layer draft head; without this flag the standard EAGLE worker runs against it and outputs garbage. + +**Reasoning effort.** Pass `reasoning_effort` as one of the named levels below; requests that omit it default to `high`, and `max` is the strongest. Each level maps to an internal effort value (max at `0.99`): + + + + + + + + + + + + + + + + +
reasoning_effortvalue
none0.0
low0.2
medium0.7
high0.9
xhigh0.99
max0.99
+ +## 3. Advanced Usage + +### 3.1 Reasoning + +Enable the `inkling` reasoning parser (toggle **Reasoning Parser** in the **Parsers** card of the [Playground above](#playground)) to separate thinking from the final answer into `reasoning_content` vs `content`. + + + +```python Example +from openai import OpenAI + +client = OpenAI(base_url="http://localhost:30000/v1", api_key="EMPTY") + +resp = client.chat.completions.create( + model="thinkingmachines/Inkling-NVFP4", + messages=[{"role": "user", "content": "What is 17 times 24?"}], + extra_body={"chat_template_kwargs": {"thinking": True}}, +) +msg = resp.choices[0].message +print("Reasoning:", getattr(msg, "reasoning_content", None)) +print("Answer:", msg.content) +``` + + + + + +```text Output +Reasoning: The user is asking for the product of 17 and 24. Let me calculate that. + +17 × 24 + +I can break this down: +17 × 20 = 340 +17 × 4 = 68 +340 + 68 = 408 + +Alternatively: +24 × 10 = 240 +24 × 7 = 168 +240 + 168 = 408 + +So the answer is 408. +Answer: 17 times 24 is **408**. + +Here's a quick breakdown: +- 17 × 20 = 340 +- 17 × 4 = 68 +- 340 + 68 = **408** +``` + + + +### 3.2 Tool Calling + +Enable the `inkling` tool-call parser (toggle **Tool Call Parser** in the **Parsers** card of the [Playground above](#playground)) to surface structured tool calls via `message.tool_calls`. + + + +```python Example +from openai import OpenAI + +client = OpenAI(base_url="http://localhost:30000/v1", api_key="EMPTY") + +tools = [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the current weather for a location", + "parameters": { + "type": "object", + "properties": {"location": {"type": "string", "description": "The city name"}}, + "required": ["location"], + }, + }, + } +] + +resp = client.chat.completions.create( + model="thinkingmachines/Inkling-NVFP4", + messages=[{"role": "user", "content": "What's the weather in Beijing?"}], + tools=tools, +) +msg = resp.choices[0].message +print("Reasoning:", getattr(msg, "reasoning_content", None)) +print("Content:", msg.content) +print("Tool calls:", msg.tool_calls) +``` + + + + + +```text Output +Reasoning: The user is asking for the weather in Beijing. I have a tool called `get_weather` that can get the current weather for a location. Let me call it with "Beijing" as the location. +Content: +Tool calls: [ChatCompletionMessageFunctionToolCall(id='call_98f772f3a0044f45b80c5ba5', function=Function(arguments='{"location": "Beijing"}', name='get_weather'), type='function', index=0)] +``` + + + +### 3.3 Multimodal Input (Image + Audio) + +Inkling is multimodal: a single user message can mix **text**, **images**, and **audio**. Pass each media item as its own content part — `image_url` for images, `audio_url` for audio — with the `url` set to either an HTTP(S) link or a base64 `data:` URI. The server must be started with `--enable-multimodal` (already included in every recipe above). + + + +```python Example +import base64 +from openai import OpenAI + +client = OpenAI(base_url="http://localhost:30000/v1", api_key="EMPTY") + +with open("image.png", "rb") as f: + image_b64 = base64.b64encode(f.read()).decode() +with open("audio.wav", "rb") as f: + audio_b64 = base64.b64encode(f.read()).decode() + +resp = client.chat.completions.create( + model="thinkingmachines/Inkling-NVFP4", + messages=[ + { + "role": "user", + "content": [ + {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{image_b64}"}}, + {"type": "audio_url", "audio_url": {"url": f"data:audio/wav;base64,{audio_b64}"}}, + {"type": "text", "text": "Describe the image, then transcribe the audio."}, + ], + } + ], + max_tokens=1024, +) +print(resp.choices[0].message.content) +``` + + + + +Images and audio can be sent as public HTTP(S) URLs instead of base64 — e.g. `{"type": "image_url", "image_url": {"url": "https://.../photo.jpg"}}`. Use one content part per media item; mix as many as the context budget allows. + + +### 3.4 LoRA (Serving Adapters) + +The **LoRA** deploy variant serves adapters on top of the frozen base model. Its launch command adds `--enable-lora --lora-paths lora0={{ADAPTER_PATH}} --max-loras-per-batch {{MAX_LORAS}}` — each adapter is registered under the **name** to the left of `=` (here `lora0`). Adapters can also be added/removed at runtime via the `POST /load_lora_adapter` endpoint. To serve several adapters, pass multiple `--lora-paths name=path` at launch and reference each by its name. + +Pick the adapter per request by that name — either in the `model` field with `base-model:adapter` syntax (recommended), or explicitly via `lora_path` in `extra_body`: + + + +```python Example +from openai import OpenAI + +client = OpenAI(base_url="http://localhost:30000/v1", api_key="EMPTY") + +# Option A (recommended): ":" in the model field +resp = client.chat.completions.create( + model="thinkingmachines/Inkling-NVFP4:lora0", + messages=[{"role": "user", "content": "Summarize the changelog."}], +) + +# Option B: explicit lora_path via extra_body +resp = client.chat.completions.create( + model="thinkingmachines/Inkling-NVFP4", + messages=[{"role": "user", "content": "Summarize the changelog."}], + extra_body={"lora_path": "lora0"}, +) + +print(resp.choices[0].message.content) +``` + + + + +One adapter per request — omit the `:adapter` suffix (and `lora_path`) to hit the base model. Different requests **in the same batch** may use different adapters; the number of *distinct* adapters co-resident in a batch is capped by `--max-loras-per-batch` (the `MAX_LORAS` field, default `1`). If both `model:adapter` and `lora_path` are supplied, the `model` suffix takes precedence. + + +### 3.5 HiCache (Hierarchical KV Caching) + +Inkling serves on SGLang's **unified radix cache**: the historically separate full-attention, SWA, and Mamba/sconv caches are combined into one radix tree with typed components, and native HiCache offloads cold prefix pages across tiers (GPU HBM → host DRAM → disk / remote). This expands effective prefix-cache capacity for multi-turn and long-context workloads. + +To enable HiCache, open the **HiCache** card in the [Playground above](#playground) and flip **Enable**, then pick a storage backend (`file` / `mooncake` / `nixl`) for the L3 tier. The Write policy defaults to `write_through`. + +### 3.6 Long Context (MXFP8 KV) + +The **Long Context** deploy strategy adds `--kv-cache-dtype mxfp8` on top of the Balanced recipe. KV entries are stored as block-scaled MXFP8 instead of BF16, so the SWA + Mamba/sconv memory pool holds roughly 2x as many tokens on the same GPU. Use it when you're context-bound or concurrency-bound. + +**Blackwell only.** MXFP8 KV cache requires Blackwell (B200 / B300 / GB200 / GB300), it's not offered on Hopper (H200). + +The tradeoff is a ~5% decode latency penalty from the extra quantize/dequantize work versus BF16 KV, so treat it as a capacity lever, not a speed one — stay on **Balanced** if you have headroom in the memory pool and just want lower latency. + +To try it, select the **Long Context** strategy in the Deploy panel above for any NVFP4 cell; the panel regenerates the launch command with `--kv-cache-dtype mxfp8` inserted. Verified end-to-end on B200. diff --git a/docs_new/cookbook/autoregressive/intro.mdx b/docs_new/cookbook/autoregressive/intro.mdx index 8183d9170..5dac02921 100644 --- a/docs_new/cookbook/autoregressive/intro.mdx +++ b/docs_new/cookbook/autoregressive/intro.mdx @@ -37,6 +37,12 @@ metatags: href="/cookbook/autoregressive/GLM/GLM-5.2" img="/cards/logos/glm.png" /> + a single honest `balanced` tier. + strategies: [ + { id: "balanced", label: "Balanced" }, + { id: "mtp", label: "MTP" }, + { id: "long_context", label: "Long Context (MXFP8 KV)" }, + ], + nodesOptions: [ + { id: "single", label: "Single Node" }, + { id: "multi-2", label: "Multi-Nodes" }, + ], + + // HF repos under the thinkingmachines org. + modelNames: { + "default|nvfp4": "thinkingmachines/Inkling-NVFP4", + "default|bf16": "thinkingmachines/Inkling", + "lora|nvfp4": "thinkingmachines/Inkling-NVFP4", + "lora|bf16": "thinkingmachines/Inkling", + }, + + placeholders: { + HOST_IP: { target: "command", label: "Bind host", default: "0.0.0.0" }, + PORT: { target: "command", label: "Bind port", default: "30000" }, + NODE0_IP: { target: "command", label: "Head node IP", default: "" }, + NODE_RANK: { target: "command", label: "This node rank", default: "" }, + HF_TOKEN: { target: "command", label: "HF token (Docker)", default: "" }, + ADAPTER_PATH: { target: "command", label: "LoRA adapter dir", default: "" }, + MAX_LORAS: { target: "command", label: "Max LoRAs per batch", default: "1" }, + CURL_HOST: { target: "curl", label: "Server host", default: "localhost" }, + CURL_PORT: { target: "curl", label: "Server port", default: "30000" }, + }, + + curl: `curl http://{{CURL_HOST}}:{{CURL_PORT}}/v1/chat/completions \\ +-H 'Content-Type: application/json' \\ +-d '{ "model": "{{MODEL_NAME}}", "messages": [{"role":"user","content":"Hello"}] }'`, + + // NVIDIA: two multi-arch CUDA builds (inkling-cu12 / inkling-cu13) — pick by your + // CUDA version, not by GPU. AMD: inkling-rocm700-mi35x. Panel defaults to cu13. + dockerImages: { + h200: "lmsysorg/sglang:inkling-cu13", + b200: "lmsysorg/sglang:inkling-cu13", + b300: "lmsysorg/sglang:inkling-cu13", + gb200: "lmsysorg/sglang:inkling-cu13", + gb300: "lmsysorg/sglang:inkling-cu13", + mi350x: "lmsysorg/sglang:inkling-rocm700-mi35x", + mi355x: "lmsysorg/sglang:inkling-rocm700-mi35x", + }, + + github: { + cookbookModel: "thinkingmachines/inkling", + }, + + playgroundFeatures: { + + // ----- Card: "Attention Parallelism" ----- + // TP only. Inkling needs TP=8 to hold the 1M-token SWA + Mamba/sconv pools + // (TP=4 can't fit — see §2). TP=16 is cross-node (multi-node path). + attention: { + knobs: [ + { id: "tp", label: "TP", values: [ + null, 4, 8, + { value: 16, disable: { nodes: ["single"] }, + disableReason: "TP=16 requires 16 ranks — switch the Deploy panel's Nodes to Multi-Nodes first." }, + ]}, + ], + }, + + // ----- Card: "MoE Parallelism" ----- + // Blackwell (SM100) runs the FlashInfer TRT-LLM routed FP4 experts; Hopper (SM90) + // has no FP4 runner and falls back to Marlin W4A16. + moe: { + backend: { + options: [ + { id: null, label: "Inherited" }, + // NVIDIA backends hidden on AMD; AITER/Triton hidden on NVIDIA. + { id: "flashinfer_trtllm_routed", label: "FlashInfer TRT-LLM (routed FP4)", + flags: ["--moe-runner-backend flashinfer_trtllm_routed"], + requiresHw: ["b200", "b300", "gb200", "gb300"], + hide: { hw: ["mi350x", "mi355x"] } }, + { id: "marlin", label: "Marlin (W4A16)", + flags: ["--moe-runner-backend marlin"], + hide: { hw: ["mi350x", "mi355x"] } }, + { id: "aiter", label: "AITER", + flags: ["--moe-runner-backend aiter"], + hide: { hw: ["h200", "b200", "b300", "gb200", "gb300"] } }, + { id: "triton", label: "Triton", + flags: ["--moe-runner-backend triton"], + hide: { hw: ["h200", "b200", "b300", "gb200", "gb300"] } }, + ], + }, + }, + + // ----- Card: "Parsers" ----- + parsers: { + items: [ + { id: "reasoning", label: "Reasoning Parser", flag: "--reasoning-parser inkling" }, + { id: "toolCall", label: "Tool Call Parser", flag: "--tool-call-parser inkling" }, + ], + }, + + // ----- Card: "Speculative Decoding" ----- Inkling ships an MTP draft head. + speculative: { + options: [ + { id: "current", label: "Inherited from base" }, + { id: "off", label: "Off (greedy)" }, + { id: "mtp", label: "EAGLE / MTP 8-1-9", + flags: ["--speculative-algorithm EAGLE", "--speculative-num-steps 8", + "--speculative-eagle-topk 1", "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", "--speculative-use-rejection-sampling"] }, + ], + }, + + // ----- Card: "PD Disaggregation" ----- NVIDIA only; Mooncake MNNVL env gated to GB200/GB300. + pdDisagg: { + modes: [ + { id: "off", label: "Off" }, + { id: "prefill", label: "Prefill role", hide: { hw: ["mi350x", "mi355x"] } }, + { id: "decode", label: "Decode role", hide: { hw: ["mi350x", "mi355x"] } }, + ], + transferBackends: [ + { id: "mooncake", label: "Mooncake", + env: [ + "MC_FORCE_MNNVL=1", + "NCCL_MNNVL_ENABLE=1", + "NCCL_CUMEM_ENABLE=1", + "SGLANG_MOONCAKE_CUSTOM_MEM_POOL=True", + ], + envWhen: { hw: ["gb200", "gb300"] } }, + ], + // Router fronting both roles; 8998 = prefill bootstrap port (default). + router: { + port: 30080, + command: +`python3 -m sglang_router.launch_router \\ + --pd-disaggregation \\ + --prefill http://:{{PREFILL_PORT}} 8998 \\ + --decode http://:{{DECODE_PORT}} \\ + --host 0.0.0.0 --port {{ROUTER_PORT}} \\ + --disable-circuit-breaker \\ + --health-check-interval-secs 999999`, + }, + }, + + // ----- Card: "Hierarchical KV Cache" ----- Native HiCache over the unified radix tree. + hicache: { + backends: [ + { id: null, label: "Auto" }, + { id: "file", label: "File" }, + { id: "mooncake", label: "Mooncake" }, + { id: "nixl", label: "NiXL" }, + ], + writePolicies: [ + { id: "auto", label: "Auto" }, + { id: "write_through", label: "Write-through" }, + { id: "write_back", label: "Write-back" }, + ], + }, + }, + + cells: [ + // ==================================================================== + // NVIDIA Blackwell (SM100) + NVFP4 — FlashInfer TRT-LLM routed FP4 experts. + // B200 verified; B300 / GB200 / GB300 same-arch (GB300 in active validation). + // ==================================================================== + { + match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + verified: true, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb200", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + // ==================================================================== + // NVIDIA Hopper (SM90) + NVFP4 — no FP4 MoE runner on Hopper -> Marlin W4A16. + // fa4 SplitKV auto-sets num_splits=1 on SM90. H200 verified. + // ==================================================================== + { + match: { hw: "h200", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + verified: true, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend marlin", + "--moe-runner-backend marlin", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + // AMD ROCm (MI350X / MI355X) + BF16 — verified, TP=8. `--moe-runner-backend` + // sits right after `--tp` so the Playground AITER override (re-inserted at + // that anchor) reproduces this command exactly. + { + match: { hw: "mi350x", variant: "default", quant: "bf16", strategy: "balanced", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_AITER=1", + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--moe-runner-backend aiter", + "--attention-backend triton", + "--disable-custom-all-reduce", + "--disable-prefill-cuda-graph", + "--mamba-radix-cache-strategy extra_buffer", + "--page-size 128", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.2", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "mi355x", variant: "default", quant: "bf16", strategy: "balanced", nodes: "single" }, + verified: true, + env: [ + "SGLANG_USE_AITER=1", + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--moe-runner-backend aiter", + "--attention-backend triton", + "--disable-custom-all-reduce", + "--disable-prefill-cuda-graph", + "--mamba-radix-cache-strategy extra_buffer", + "--page-size 128", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.2", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + + // ==================================================================== + // Long Context (MXFP8 KV) — block-scaled KV cache shrinks the per-token + // KV footprint, raising how many tokens fit in the memory pool (longer + // context / more concurrent sequences) vs the default BF16 KV. Same base + // command as Balanced + `--kv-cache-dtype mxfp8`. B200 verified + // end-to-end. + // ==================================================================== + { + match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" }, + verified: true, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--kv-cache-dtype mxfp8", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--kv-cache-dtype mxfp8", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb200", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--kv-cache-dtype mxfp8", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--kv-cache-dtype mxfp8", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + + // ==================================================================== + // MTP (speculative decoding) — Inkling's multi-layer MTP draft head. + // --enable-multi-layer-eagle is REQUIRED (without it the standard EAGLE + // worker runs against the multi-layer draft and outputs garbage). + // B200 verified end-to-end; H200 from the same validated command set. + // ==================================================================== + { + match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" }, + verified: true, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.75", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.75", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb200", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.75", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend flashinfer_trtllm", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.75", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend marlin", + "--moe-runner-backend marlin", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.78", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + // ==================================================================== + // GB300 BF16 — 2x GB300 nodes (4 GPUs each) over MNNVL. The NCCL_MNNVL / + // NVLS / CUMEM envs are required: 2-node NCCL init hangs without them. + // MTP on BF16 requires the v3 MTP checkpoint + an SGLang revision with + // v3 MTP support. + // ==================================================================== + { + match: { hw: "gb300", variant: "default", quant: "bf16", strategy: "balanced", nodes: "multi-2" }, + env: [ + "NCCL_MNNVL_ENABLE=1", + "NCCL_NVLS_ENABLE=1", + "NCCL_CUMEM_ENABLE=1", + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--dist-timeout 3600", + "--moe-runner-backend flashinfer_trtllm_routed", + "--attention-backend fa4", + "--disable-custom-all-reduce", + "--enable-torch-symm-mem", + "--page-size 128", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb300", variant: "default", quant: "bf16", strategy: "mtp", nodes: "multi-2" }, + env: [ + "NCCL_MNNVL_ENABLE=1", + "NCCL_NVLS_ENABLE=1", + "NCCL_CUMEM_ENABLE=1", + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--dist-timeout 3600", + "--moe-runner-backend flashinfer_trtllm_routed", + "--attention-backend fa4", + "--disable-custom-all-reduce", + "--enable-torch-symm-mem", + "--page-size 128", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b300", variant: "default", quant: "bf16", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--attention-backend fa4", + "--page-size 128", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b300", variant: "default", quant: "bf16", strategy: "mtp", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--attention-backend fa4", + "--page-size 128", + "--moe-runner-backend flashinfer_trtllm_routed", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "default", quant: "bf16", strategy: "balanced", nodes: "multi-2" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 16", + "--moe-runner-backend flashinfer_trtllm_routed", + "--attention-backend fa4", + "--disable-custom-all-reduce", + "--page-size 128", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b200", variant: "default", quant: "bf16", strategy: "mtp", nodes: "multi-2" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 16", + "--moe-runner-backend flashinfer_trtllm_routed", + "--attention-backend fa4", + "--disable-custom-all-reduce", + "--page-size 128", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--speculative-algorithm EAGLE", + "--speculative-num-steps 8", + "--speculative-eagle-topk 1", + "--speculative-num-draft-tokens 9", + "--enable-multi-layer-eagle", + "--speculative-use-rejection-sampling", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + // ==================================================================== + // LoRA serving. Prefill CUDA graphs auto-disable under --enable-lora. + // Set MAX_LORAS to the number of distinct adapters served (1 is fastest + // for single-adapter serving). All three cells verified end-to-end + // (coherence + trainer-logprob parity + throughput). + // ==================================================================== + { + match: { hw: "b200", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + verified: true, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + "SGLANG_EXPERIMENTAL_LORA_OPTI=1", + "SGLANG_OPT_LORA_OVERLAP_MAIN_ALLOC=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend marlin", + "--moe-runner-backend experimental_sgl_marlin", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.80", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--enable-lora", + "--lora-backend triton", + "--lora-use-virtual-experts", + "--max-loras-per-batch {{MAX_LORAS}}", + "--lora-paths lora0={{ADAPTER_PATH}}", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "b300", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + "SGLANG_EXPERIMENTAL_LORA_OPTI=1", + "SGLANG_OPT_LORA_OVERLAP_MAIN_ALLOC=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend marlin", + "--moe-runner-backend experimental_sgl_marlin", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.80", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--enable-lora", + "--lora-backend triton", + "--lora-use-virtual-experts", + "--max-loras-per-batch {{MAX_LORAS}}", + "--lora-paths lora0={{ADAPTER_PATH}}", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb200", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + "SGLANG_EXPERIMENTAL_LORA_OPTI=1", + "SGLANG_OPT_LORA_OVERLAP_MAIN_ALLOC=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend marlin", + "--moe-runner-backend experimental_sgl_marlin", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.80", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--enable-lora", + "--lora-backend triton", + "--lora-use-virtual-experts", + "--max-loras-per-batch {{MAX_LORAS}}", + "--lora-paths lora0={{ADAPTER_PATH}}", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb300", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + "SGLANG_EXPERIMENTAL_LORA_OPTI=1", + "SGLANG_OPT_LORA_OVERLAP_MAIN_ALLOC=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 4", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend marlin", + "--moe-runner-backend experimental_sgl_marlin", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.80", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--enable-lora", + "--lora-backend triton", + "--lora-use-virtual-experts", + "--max-loras-per-batch {{MAX_LORAS}}", + "--lora-paths lora0={{ADAPTER_PATH}}", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" }, + verified: true, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + "SGLANG_EXPERIMENTAL_LORA_OPTI=1", + "SGLANG_OPT_LORA_OVERLAP_MAIN_ALLOC=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--quantization modelopt_fp4", + "--attention-backend fa4", + "--page-size 128", + "--fp4-gemm-backend marlin", + "--moe-runner-backend experimental_sgl_marlin", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.85", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--enable-lora", + "--lora-backend triton", + "--lora-use-virtual-experts", + "--max-loras-per-batch {{MAX_LORAS}}", + "--lora-paths lora0={{ADAPTER_PATH}}", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "gb300", variant: "lora", quant: "bf16", strategy: "balanced", nodes: "multi-2" }, + verified: true, + env: [ + "NCCL_MNNVL_ENABLE=1", + "NCCL_NVLS_ENABLE=1", + "NCCL_CUMEM_ENABLE=1", + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + "SGLANG_EXPERIMENTAL_LORA_OPTI=1", + "SGLANG_OPT_LORA_OVERLAP_MAIN_ALLOC=1", + "SGLANG_OPT_USE_JIT_KERNEL_MOE_ALIGN=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--moe-runner-backend experimental_sgl_trtllm", + "--attention-backend fa4", + "--page-size 128", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--enable-lora", + "--lora-backend triton", + "--lora-use-virtual-experts", + "--max-loras-per-batch {{MAX_LORAS}}", + "--lora-paths lora0={{ADAPTER_PATH}}", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + { + match: { hw: "h200", variant: "lora", quant: "bf16", strategy: "balanced", nodes: "single" }, + env: [ + "SGLANG_ENABLE_UNIFIED_RADIX_TREE=1", + "SGLANG_EXPERIMENTAL_LORA_OPTI=1", + "SGLANG_OPT_LORA_OVERLAP_MAIN_ALLOC=1", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tp 8", + "--moe-runner-backend triton", + "--attention-backend fa4", + "--page-size 128", + "--enable-torch-symm-mem", + "--mamba-radix-cache-strategy extra_buffer", + "--mem-fraction-static 0.87", + "--swa-full-tokens-ratio 0.1", + "--mamba-full-memory-ratio 0.1", + "--enable-multimodal", + "--reasoning-parser inkling", + "--tool-call-parser inkling", + "--enable-lora", + "--lora-backend triton", + "--lora-use-virtual-experts", + "--max-loras-per-batch {{MAX_LORAS}}", + "--lora-paths lora0={{ADAPTER_PATH}}", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, + ], +};