From aa16e0006ae92f5ae3c65594be576e02ac62b9f8 Mon Sep 17 00:00:00 2001 From: "yuze.zyz" Date: Fri, 25 Oct 2024 17:28:47 +0800 Subject: [PATCH 1/7] wip --- .../model_providers/_position.yaml | 1 + .../model_providers/modelscope/__init__.py | 0 .../modelscope/_assets/icon_l_en.png | Bin 0 -> 14426 bytes .../modelscope/_assets/icon_s_en.png | Bin 0 -> 3320 bytes .../modelscope/llm/__init__.py | 0 .../model_providers/modelscope/llm/llm.py | 518 ++++++++++++++++++ .../model_providers/modelscope/modelscope.py | 10 + .../modelscope/modelscope.yaml | 38 ++ 8 files changed, 567 insertions(+) create mode 100644 api/core/model_runtime/model_providers/modelscope/__init__.py create mode 100644 api/core/model_runtime/model_providers/modelscope/_assets/icon_l_en.png create mode 100644 api/core/model_runtime/model_providers/modelscope/_assets/icon_s_en.png create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/__init__.py create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/llm.py create mode 100644 api/core/model_runtime/model_providers/modelscope/modelscope.py create mode 100644 api/core/model_runtime/model_providers/modelscope/modelscope.yaml diff --git a/api/core/model_runtime/model_providers/_position.yaml b/api/core/model_runtime/model_providers/_position.yaml index 89fccef6598fdd..3a40282909309e 100644 --- a/api/core/model_runtime/model_providers/_position.yaml +++ b/api/core/model_runtime/model_providers/_position.yaml @@ -15,6 +15,7 @@ - groq - replicate - huggingface_hub +- modelscope - xinference - triton_inference_server - zhipuai diff --git a/api/core/model_runtime/model_providers/modelscope/__init__.py b/api/core/model_runtime/model_providers/modelscope/__init__.py new file mode 100644 index 00000000000000..e69de29bb2d1d6 diff --git a/api/core/model_runtime/model_providers/modelscope/_assets/icon_l_en.png b/api/core/model_runtime/model_providers/modelscope/_assets/icon_l_en.png new file mode 100644 index 0000000000000000000000000000000000000000..123d50cf4713d13d2e49458c6fb56e04fc623eec GIT binary patch literal 14426 zcmeHuc{r5s_dgyQ22SxACr|*BiKfb@~`nj%|xy}7N&pG!w=RW6k&V43O6{2vFh?WQo3+p0CQ5K4Y zg_{YCF$Cv<_smLeDc}R!0jh8hE3boo7Wl^%t^+bxQNiK@#spZn*tA%9N0$IENo=~) zu{`#5taHcT<6vP0SYYA)x&{J#9{q&_ucLcTKF_6M|GFCZE*0nZ)wr3d=YEf|Fh{rb z<_B;BAB0a7bseyh|sEMCb&R34Xun>p*wnvZk~?U*WCvxCr@fqs}Ouc+nY&nLTqeE5r80t{{9I z0w!OqUvH18UNnU5yaioKnxpu%%5kHrxVn0Djq9piBXL_onb|T%buMGoDv_^C%xsfG zoX^QHiuU;K1|r#@Ttx|7QfVygleZiQ6LRORTF&TeJ13`V$MSk+{waz_VYjmp+b{SH ztp&N+RKjChzO*p|m)G(61t}>fIX^2mqS+L!m9}KF(RnRnTcv$is@2$fd!4vxcaFx* zXc1X^#fq4qs&GNFTFFn~jr?y~Cse7gkv6zbq*@Q@pF;Ze*g8dQGIbUq*RbEhJ0r)WDgex}rDtCILn6Ov z`bFTdX3ATx(Eo*(R5P52$P!tZe=R~9Ni0jJSM#b^?6qD1jliq*-F~MrSnfAdeF7#` zcYqK|1TA+$=rQpvp^|zeim%k*)ib=0PIKRDYVr0t;(n_Ztj+Eo~WbM4fGLdcB6veoFTPA%d0x`9VsM*`x6f_g~)CeG`? zO44t;`u$CuTt};n%w+hRIMt5CX}5y&H*snJb6_?V@qfwl7lCwKMdi@16L^2&MLHXp zLxSU-|KR;}J@SO%v9t~EiT}n66qw^e-dg;_6#ftl6TeWwr5f+Bl?$hkIj_qPdJ zsxb|Hw)gMbNwj3}*q>|~tFCsbg)aO|3tgovfqtTbjnBJB^FPQptqAPLP#;EP4#N_t z@8($*8^$U-vp*gAWWS{{M)&%|rux^Zg^vTJ4?AtHE--$-TKR5bUS}c8K*ilP)%ZOT zCuyIhol)WZHIC6x;rh;gz5O~+5}1oF===R#gW+v6ey=d0;yoR2#zaHqQE--^b>rly zhr5{YmErzJrmcFp8~DMwjFay2*`Gklcy7d9BEMvk<%%sT;T6d$o~gYj%}(tLM@%En=on%ENkTDo2Q78e<2F<6c5zOuBi!A*$zT@8z`u$=xKppbGhfEWrkJ@iuVfN z%>8f>%MFGsB!nkFelxcZi-TOM+<|3>CYx(G?I?b*=JFv3+9?6E{Y4kpWq|HSh@gyr zvx_5JlPgdo{L5TjE><9C**}=Sx&v_C>N9J(l6$`i68QR#F}Va(es~>5ZOXvoa$z@( zYPGwiine=)na_k1ZcdwiCumCL<%@q#tvVbAY%LqAXy#+XPz}YQBR`iUwP9)8)%Y{l zzU*yO^$V)*>ArgDADo8uLGrMf>!uO0YdBHVzwO4i6Gy*gM)2mB%cmxJW~TPUfLC4o z0R5XlQt%X&w=XEW{$e<=bET{a;tqJp_a$d5-+*UxdX}hUGHV`UgEGUz{Om z%Z2iV*yx=o{YHXEPDzw04@8oB^a3VrqBdtzDtkZV3O&!?6XU+-&MQSi&VMc-fJQLXVSA~|KGuG-fO zv8427KCCkEN-%?HC-`E%9XU&uQ;BdH$N7r;c0IL#<3ooC_&gzJdFSLLR-R`SG<+aO zr)L;WU0E3_@&MD(-7csn&E&!gc7J&n1x<$;&$EA-+j zCvD*g2{m%*xx`4g#20pX)p^?7PB4ihuLD4&c+;a%=kSwCNQ4thz@7Cr5?DY0I#i$f z`;3uxVG3S;KqSAXdPNDFt3>N43j@^96BaFgTZvoExSD=Jqpj@Ynj!IIz?Y}wfuhLb z)k6$s1Io4;Q*&Y>Ck{kQZN2XPlk;dG^5RPHc^TM5`6?t`S$pSiBHkvstqVg25#|Mz;a{JM`KY?2V~8Q?gxfDlq3 z!s0yrD+EN5Gx2MRS?S&TbG^R^P+yZavBW)p|8F)1<{;%XP%nP==c3`{pLc|E+iY9= zyu+S<4p6P5i|1u{n3Hax2N1cE_S^Q-V6A3^t z8-GL?@|U^r`@YSc>$7i9+^HWkF4=rVG_5IwD{bx_S3zT+dfG+azIPOIJfFd}OkTTB zE^9hyfk+&^?l!uCJCGz)mbsxHxvrdr$t)az56!oiX>g91lwQHD0vF4s^Xln>J^1f6 z@41paG&8EAooR~w18HY5VBeT}on+mv_}{YklLk~xgqcni{EKZb0w$94Ki+rqlSZmA ze$pwNVyC=0v@nkOy!JX!>58Fgcebv-lBPLf`UZ!=?jtKIO_cg-@cnccWB!i|{ zQ{A)_X^!J@GPIMuC~A^76u|G%@_oxusD^7=t+A?e^WYnii|3RW_ue^OJ3Lk_`8!Y(Nw>%b^0iOR6Ki(c)1fT$fv-VbeX?_7@f-4%-7E26 znhMQ_66g|I5!TKP z7!GaWTaQ~sGNp5S<}T-Iuqt{LelS;mbaT&-X){~ZsJJgvm8S3CanGAbHcea3B{HK$ z(p&he#lqX29hDz*hikXYFP<0VoJTjt%5cHIXJ4dC>+bgh&*p9lKucApQZ8{sd@Isriw2sZ8Hir&O0MS{@#~~vOSXeUL)2TmEC((jAXtvQGijl zPUI|2PTKi3lBms#H!8A3l%|3zn!}>rXMW&X%<#-B{PJX$UwFHG5|`e0OFnX&4Psq#-g$hTQVg8;2;bm9>u!G|3Ljfr)7`!3h% zD{d*;6q&b=XJ91cTsSjiIO5*M={x#jXLuraBANvnHuX7d__m+Qfv@m^Zdd#}$?KKov5D&B^Sgp2ztHM?4ElQY7?cbkLF z*rgbUGPx|*))SUAudz$s?x&A+)|7S%f z%XDS{r=u1&W1jU(D5rddMX6;q`VR<2AXf}bwK)G@CeTRC+V5nk#7*n!qZtyAD(HzOoDGO-?tnYDKs3x#8eMzds!>6Ty(jI-1zvJQjrND? zPfxq4o9+d)WEl`Oex94Em+p5VicZJEA;FEOdpW{q4W$~$A=5gHP;}gXucdp5xFKz| zClp0eN?z#Q@@C!L!*=*Fng zumIcTSTU#WN#2A~?=Fa`p0lIdJFxOai|W?`(YafQuAU5uPBQhi4)Mn8>k*6TP3tLI zVw|r*%2WI%NC!X1M?YoQb00764iQOk@gnZr)vt}+BtDVPY zJj&o-C_VZwWoAA0+g^`pqa86xG8jz1ZnP|Q^C9}M(z2KLDfuQ4C}ahA{ws+1#ZxKT z5-yu5&#lD=Tqm38)fO@0(*!9~%qj?-5V%8id*SUmyTsv-8A^|<19H->S)$-PpSSBj z+t+I8#YJd+X8yS@PT|Q#9(P5zci7_xx1>J@!aCd9pMnuFVP>!hStr|7pi-`}YW zn-*b-(d>3;)u|=X%Zguc!fR~KQO?fD#QdN~J=W`M9X8CX89+w%IGAED)R^l%-pst+oGDajDqIJKC zu`O=w`6geyCuyuhZ&C2op7qviH@HMQOf&KQNMTp~rQ@ zP{w>X9YTA{!XW?gcyzBmh4rKIHprghKEi8vN33s{Q;7zbH0vmH=`B;Cx%@BE;o3|T zS`U!WqIF8zvhMfwhmCFFb8$X+jM*6Byh+Y1S!~=K0O4qC+Q8(WpYr;e=pN{GL zD!cV)>6g&Z`}B^VyE|jJMN1?&d#MjmD2#@$!V^TeWF{fVk4|s=-i^D@o zf|c^_zlO$ z#=M{JMd>4V@dG)3JeNI%Po!Db{A&F20^sk&DFi2v%5d3V-os^{(haZm_UlKPg0HdHQ8GM(69v8zkAXq7)$%U@&}_V=$}7ySfTplbgGU2@4Bz zbapF{nu?eET{k`UHWOuaRnJ*WBN6WRXre?ok+T8(mhto%BiCYN3`!oUi5s!+R@=Qe zs^%d;(l^E{6}@L>RKW1cEr@-KSw~`ggu!(0{Yp^wk0HV7j4|~27>`wvpn8l)&YhFW zMQS1L9DC=01XKz>Q*h|0c!*h2bkyh{~JV zA+v&+o^5w81TmB9V2kd~zC%DHZqT6FIg$|P{PS~t>^7h7E_qDuCSlT#1GV%D-lTH& za%&~nwr>hsaH@)pp3qQm#5y>q1;-!E8H4_ZsvZ-&AN5HRES{|L?tTYGMU%RferG-SPV zGePyGS;GQ6;^|`OM@61=ew(^VzIdFAy0RK?tdoKALIlGgv}F+BoUR3;swt!S`Uw(n{uchvacuKV2^JQgbv9TBIa{mh7RU>2ZY%8=xImV=8 zy39zA4WHg{M9<>~0<_QX@QGdATKyPMGyTK~bTTMwQl3G#dd`G#T>kO3jk6!bnd{B& zz?eLv6)$*pe&ok>B^-v$_`t#;FX>ftKwvOj;5)|x0(Aigg1fV$^`=DK`*45Q+;*_I z7t*C1r_CWYpOwFVwUVDG+6t(yE24T_ng$`2S9l)R*0GjPetZ7Nbg!;@YwX(yk~@FU zyq1gYxZ?89^>7eGNT_})QO`@Avonzh4CN|P0U*3xXU6?s0SU&GA(_ErI?r=ODL}AH zcpAVN%S266#p5?;H!e+Ku`Cb=@f+>z3U)4v*p7WoPw{qdAg(FCtBPw+P;at&P8saQ z`zgmRcyjNhRQ3x)F&nusJoelg4*Y3Bhd1_BijnWPX4sX{c`?ospq0576`8fQ^JHS#{0n7yWp_j*VvWfpnY7W<(G^yq zaD(d3jktw9ujI%#_v!?(MPo-*wid{(@h{#XR zNmsZ?7P52{bCK5oJ9Bd@t6}0m3BaD!Yrq#`HptGfoiq;MB``~~ zU*1b`nReLq@aFc0dMs6c;xJ3jN_ob-W9+u?S+L3~+V>dsJULwOonCP0-Q(^LKVt${ zKvV8euJm@4=lg+~?`P-hK=xMb?xU`*G*;=Xl8~7)H#<8dpypx{8hmL``1YO^2wZKO zCLkPrsYSbfhBb0jgyZ}eC$G!Kj*@y$1aYyDXMp|(_ZOkmqCv3?W@FhNX%|;pAvzVQ z23H0A=`I~vgdh0I%Gbggonj%*nD`XEwx3h2ktuBFD>}I|^P#{q%tMD7pJCzImy@(+)3Aqe8HdC64!-#MY^Y*yVSTC{S7;>EbQr|CMNAl{v&3R=OCjmH12~;!^A76{WiN zIQ52&Yy>D<`A|`rq6nFF0vM$m2*0=ahY%+`)S7Qo3rb{LntZ^1XT^@pzmNcN zFSYzZ^Ik3t>H>y&P>d&+PCh+g+i&hrf)_f#mZ{=ga!i*-e(G82jPgY7Q}P@cb|63o2_*{Rg(|`ghHxDb2bz(vv}6bmZJMzTqiJrjXeD6 z#5izBE4i3)H}2)c<5q$Gp=|Q*SFVS3LAwix_UJ(2(mO2KJXHAGgNl2ybw+*wfXwDZ z)@W_@b2gN+F2C=e%JWbn52U!%1NFK=-U_BM4fJwb%Td}*)$Yt~w&f7v+yXMx%v zp<++Rm{Ce^uil(34Su^5S*RLf*G8#NDT3j$AZ4UWZJN`mL-WHTqIVu}i%RyIE>CHJ zJ3g$KO=eBgntw9NzrXK7;Z;_hoE&zs`(>ZbE$;&5paZYjmqiPGmFL3M>G)i?QN8$W z3H?J`8$C&fj)FJlm?u!u?`-THcQ)J;9?nkNx_wNb%a8ieO1Q05U8EJj;<5DhW3k8r zf>*=51@WAB|JfQiKHWJlWj*FmL)o1*zM*|&J8jI1zBJ?)MezC z4*KntnY;EC)%dfX`Ch~xJp33;Su!j(hA8?eJV;>i`o&19ag)~jMgNWAmlDpq^1D-< zfzmf|s~`}hASd#2$!0J!^~T1ZO*e;lW*=5bK;Yg=3y|dGrRFIg;^h5d2cPIx?9pem z2HNJqQRfcY?WZ;D0yzWOO%qrI4Ra%voNJjK4EIeO&sE*4qr+}3%=6lA=qvQ_ju$_^Z;)PCmK1}@YK|bLj_8S z4TM%j`#H=eUrmiIk26NiX8L{0<#@ScS0f3uWR)gcg=p$Sxq3v>Ow)mQB z+YF*x*MMV5N@v=WW5QGf&h-A;`)@iNykcP5=0u^m`@)c`9boZ5WnJ7B7N(ef6f2>A zL`v>do)>=B#*_OzeeU;sd}f|bUCz)v76=DU*Q0s7#zOCf3`kz#n#e@33y%>)bUlWu z6`7dRB$Zr#Q}--qRyDv8!0#Tkm+xHimoI_$OknQ84_vCfZ#lqQ#$!v9cFmXchENI4a+vo(tkHL|JQI)mE$})%s$( z6ZzxLRqMxg$O&4ip)WX~(%ckbul^|E+WocONHej{2b(Uz32)&TNJB45OBFAqFSJD< zh0yZwI~4#V$zKQ;d&-c3l70M`9v1N#tYeH7Bh@)##6&6V<)tmoYz;R@Ce?Og30+KL zpxXARxcd^Mf6lkY&*LgzW(}j0ySoi_bjI*D*A+3)PFaH;ot{+Hd?Mbslm1JhUe*-F*gIIfTy&m`GW%%O5n9Wk@A+pbAaf66HpAF%xTY!9z;!Wa=QgR+8wH zvv@^-dzXp-v)fb8&qk9Ik&@&@v#l@1AG~jviVsc?hY1Y2?N&XSkAI*v$`%7(G|1Z9 z+SIERwy+;1e63Z2evCw`Y7$}QL9P84Mz((hp6|zC$V4ohI%MZ{feyw4LRq@6;}Nx* zG28tgDkDNgo%e%Z%S0w}4UXfIsb96$6kXh2M>9Tf-nyQ6_#kBN-K^w()!{|Ao_`Xx zT%!`x4!mvCwi($>vZ?lL8!9OwFBAAUmD<;Cq%B8$AeqtKs+PHRoHD7FXD3cB zH%26kd6%b`moE+g&2X}``T0<8k$oaQ5O8JI0Me3bpIp|l_nq&dlu3LS8M88M0J@@7XLAqfBM zUkCzDXhx#U;MKzx!{a}N)z>$JyFDSuGd`(H0%8~Cxo?N-N!7$ZjwucAo7uYLxGzx9 z@nhP{I@QOx!E5){)#@jTwv|d9kz=+Vrfu9)uibL-`?)rP+dOjm7^tOi8i<5_vd`D#WN68vi+CI;S8A)E^kwzx&cwwv-k;EE1wg7CQeTk=V-;m%tWD%VXLNt*JXjp02&PP`sDD`$08)mRkx88$rCS;@365*r z8XBSvRCd?A0dr+J$Ba)9L)+?E(rxFMVY9^Y($Q?1T0)dE_|VhZN8232ds1(}CJS$E z$&b#^Q1fwM2T*0e47(^BdfdlmKRDl|g^E6S$~F)kGaeb+Vsp*=IL4jeO(vYkL~z)% z@N>`H-6WhQ3&Ea-wijzfR;{H#+2E098^DQ1jy_Z0=QYz-9jYsWZTEl`DZ4atI%Fj} z4VM_-ttvsIt(1f_hg1Hem9mb?Pto8}HiQB~fO34q5e3;Fi-C}ybPxKG0m)RUzA259 z4Zq4DQ139U1S|?Qkutbfx)eRB5v<+JNv1x?`|(#<;Rv@&6K-re9kfFNw#I9%x1v=z z#pO7E*;^{^dv_{h!$g4x4nK)XjI23GU`1;!RW3IN{MZ^R6{Ra00Jsc7 zrvEB>Q7|(xUY%}EGFWo#HV_T< zKh81^TCmYxtgbSzgMeCqRF4WLE(6hK>Y6mdx!h**l_CMrQaU9dx|wbL)OK`yWW+hP zt@#}%T~X{2WwrEQZOurwn=r|r^i$d~Q)3?)V=ifU`*YklP92H;j({UnVm*H%y{ydm zG7&%{07s>(FQpk}2L}O){Q@Y~>rmF+b^aVNWny_2dgl3;$I57_@dT_c3%QnmHzb^d zhASMYGOr-0;&0+3$3t@du95!{Q(U>j z72mirkwJO&xQEwoxd9L>EbGN2q6?hDx0%<{czuf_W{rYd;m_XwMuH~nkBn2@eUq^b z2i3QBzRAa#AJQ@j9Ir72$E!InU|RTwWLtM%wZP_w5KrrEBp|rii9*SAr|)m-rDrY< zR@it)#@eSV82k}hgG9*GYq-dddgWKNjTh^J4#OJ0 zrU-F*)q0i}$mI3yK>%~$Nlfh4nl6+AoML3|wjA zSQ8^VXD}0E^_bHG$tM~zB zsKZYX2GH%%X#IXCNKK(iC~q8LGTl0Iu%;vHA)D40d%SaMr^V+iZ@-!v3af;K;jQKQ zc^#vrF7rtBf&sNl8$ZT1H)2Bk8|zS>{?Bxd%9>|c?ePeq%0vLlUU;&v42BOe@gtGb z9RGC<;6p*cDJBw;(D-|ur+5$Gp%6H5DkR(ccP;$n!t2?<*+-;(5AgGZ|DX#zQw!)i zg|qz({KXU-05wC5P$3rA2ePzmn;8i1bLxcsaUTVTv`NlaVH$>$_L)>m+=JfhC2uDl zg?OWkX@G6{dvO{^KMn~RdF*(c-X9Bz=>^V^z$YQ0QsI~!TY23S9!*&BQ{Nbz_IB${ zpxzu%6%rQ+58?7;w<%TQcCszzPj0E%SVL>g)c_~ZQ-+PKS}4Nrv3nBlD!Sbl6?|Mq z+`3&D-7-w)WeY&cDy$#v|JV}rMAUFeZiU7aC!Nr*1enCSg~6VCe6{a3C+tZ^s2##> z7p=WA%|=VOJmuzbc8 zyE5jsz_ivffPbp-#LNT_L+w6TfSPc$>nH0y<2A5nF zD=h`V1V7zKm~isS+xgboC(%Fd!w~}s33x#GTSY+enoQkq^p@j=Blc5G@Ob8#TVu)h zzSqK0E~UytHue@8RBwvCua)P`%cN6AfAv?Z)a9J74YF-%D}-hm@2fugsnFf%-%uu_ z@_R)`&6dx=qfCousOn75OyQTA_#3=W_5R0Oszr z&SCrDZk+e|8ZDJm#b1V~rP{#qYXN#&HED(e4gU!50Jm_8to-rSS3sA5ns`#IFz;HY z=ad$Z?5g&4wrD?hBEXeP^=|CoyI||x+gHPgki^esw6<=y-Q09i9zX#IeTZ3!~Rg1A0Q0uagRY;pT;pNQ6e~#dOQQ{%( zdX=vS_SkI*NjIENP?taKvZ)rc?i=ErFV(-EmM{2m$6MSkbMcl%b&@^ewE6RI gZYIZY;_w_DDaoC=iwj{#ztIE9L1gpp8UOQt0KGBY1ONa4 literal 0 HcmV?d00001 diff --git a/api/core/model_runtime/model_providers/modelscope/_assets/icon_s_en.png b/api/core/model_runtime/model_providers/modelscope/_assets/icon_s_en.png new file mode 100644 index 0000000000000000000000000000000000000000..8b2f2cb66f5319dd989259f00b1d432d4ab2a72e GIT binary patch literal 3320 zcmeH~iBl6<9>+T&N&_m5%qXWcA`B{^3@C>>B!UFO&3Hrvg1Q4T%2mJ;1wzPBAS#Hf zD=-*fS6PtAB?RKdVNhl`l`A-E0%Ug(5DW{5hD7(5*_o~V3%0hV?5fxA<^8_h{j2vr z^?ko&`}Bs3g$7A{hQ>>`t&m)P9*KB1l#wLi$zX3%%;fgQj}s-@RHX;a4@xl@fnb z6F8-6zm1W8)nn4;l}Dh{i1j!ek09&JRgK^oxJiZ}tMonHDEpa&nUTKp+(Y{6vUzc# zhUGfS;FFw#A)Z5xlqf9Qt*xV(*mxPWtEZJjTq+*T{i2=Rcv`1FM09ck~M=3gZD&ZiVNFg#EehLhQ35z#eg zfZ@14?K4+F(_^vM)M^a7QCXvgF)Y_f!k7rpaOCzOMCgLF1XaN}S&9 z_4ge_5{g;+kg+R`I!gF_SU7>Q-A*upVz~eJY}_)+Om0eMw{{abVd1fTIykx1oGb!qfNOaGMjSlWo_Sl)RsDb!nQv%uTWu(|0W_U7R{E#Hw%2vgRsgOXOk! znpPWe3Co5gV-dVUJd=71EwQYUh z)ssee#dR*$)t^m!>*wlZo(OwlZ-b)-jM_QalRVfoC(}=p|KAHjmnd7gecY792(#wA z<=D-O+@#mLQ^zHSN6HrJMO34QIS(Wv*>rhK^KGY$cQ*~ryqt-n8`^}nOcWRK7n0@w zXkjJkOK%3rLrjKElmn?rlFBlPU>&QDo>@+tUTSX!E>TXue{=pY~WEyux`cMl!a3UIA9%0cLXL zoE>19un9+b8UJ;^{-7V=d5gmT-Q}5meo!B z1PzIaf`DMgvV8)Omdeu4H+fbb{9-QTV%jTZl0uoQy;3Mu zN_AUWSQ0A3y0-2QUh8{1-qd_x#KzoaUdZawi>|hQSky7{FugeRRK`S%dRuz)fz~XC zzDGRXsPiH6)|kknbS;Varwzr-uD`f=TgYU0-tjP@gT*~r#kL?V?LCcq6!MCa$;A&n z^>Yrk88ugTemthr9RGew#*9qOS(KJ!K28`^ag+`mh5hKh2~NlYHm)}iph89+konHs z@Dea_9F9$IYL1Ey6y>bBdddyxxB#Ho8mkW81RRK^-_`+Qx5GL2>81aZHEosC17_I7 zF4+&l2&tO7Vz@3fg`UBP4l#~4I-;hY!i>^kHP`DFJCBNTXQV=3=H$o*mv&*%gj9R{ z%WKZxR9@W8%(;aTa50NJ}dWfLfDm@6r)b!5_K42=@BvPAOUNux(K%qpyB`-8@%h3steWMMyURDZk&XD@aM7d zio|%i1+eV~82^qFzU7zDcT3cc*;KE6U3l0eN><0mqIWu8y|iZY_KaG-{Jp}oW%gt; i(e}Pz>3I-K>0)+gcvrAJzen{C!gHUmTlL;B&c6UW29fFj literal 0 HcmV?d00001 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/__init__.py b/api/core/model_runtime/model_providers/modelscope/llm/__init__.py new file mode 100644 index 00000000000000..e69de29bb2d1d6 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/llm.py b/api/core/model_runtime/model_providers/modelscope/llm/llm.py new file mode 100644 index 00000000000000..bcd53b0f2f2a4c --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/llm.py @@ -0,0 +1,518 @@ +from collections.abc import Generator, Iterator +from typing import cast, List, Optional, Union, Mapping +import requests +from openai import ( + APIConnectionError, + APITimeoutError, + AuthenticationError, + ConflictError, + InternalServerError, + NotFoundError, + OpenAI, + PermissionDeniedError, + RateLimitError, + UnprocessableEntityError, + Stream, +) +from httpx import Timeout +from yarl import URL +from openai.types.chat import ChatCompletion, ChatCompletionChunk, ChatCompletionMessageToolCall +from openai.types.chat.chat_completion_chunk import ChoiceDeltaFunctionCall, ChoiceDeltaToolCall +from openai.types.chat.chat_completion_message import FunctionCall +from openai.types.completion import Completion + +from core.model_runtime.entities.common_entities import I18nObject +from core.model_runtime.entities.llm_entities import LLMMode, LLMResult, LLMResultChunk, LLMResultChunkDelta +from core.model_runtime.entities.message_entities import ( + AssistantPromptMessage, + ImagePromptMessageContent, + PromptMessage, + PromptMessageContent, + PromptMessageContentType, + PromptMessageTool, + SystemPromptMessage, + ToolPromptMessage, + UserPromptMessage, +) +from core.model_runtime.entities.model_entities import ( + AIModelEntity, + DefaultParameterName, + FetchFrom, + ModelFeature, + ModelPropertyKey, + ModelType, + ParameterRule, + ParameterType, +) +from core.model_runtime.errors.invoke import ( + InvokeAuthorizationError, + InvokeBadRequestError, + InvokeConnectionError, + InvokeError, + InvokeRateLimitError, + InvokeServerUnavailableError, +) +from core.model_runtime.errors.validate import CredentialsValidateFailedError +from core.model_runtime.model_providers.__base.large_language_model import LargeLanguageModel +from core.model_runtime.model_providers.openai.llm.llm import OpenAILargeLanguageModel +from core.model_runtime.utils import helper + + +class ModelScopeLargeLanguageModel(LargeLanguageModel): + + def _invoke( + self, + model: str, + credentials: dict, + prompt_messages: list[PromptMessage], + model_parameters: dict, + tools: Optional[List[PromptMessageTool]] = None, + stop: Optional[List[str]] = None, + stream: bool = True, + user: Optional[str] = None, + ) -> Union[LLMResult, Generator]: + """ + invoke LLM + + see `core.model_runtime.model_providers.__base.large_language_model.LargeLanguageModel._invoke` + """ + if "temperature" in model_parameters: + if model_parameters["temperature"] < 0.01: + model_parameters["temperature"] = 0.01 + elif model_parameters["temperature"] > 1.0: + model_parameters["temperature"] = 0.99 + credentials['mode'] = 'chat' + + return self._generate( + model=model, + credentials=credentials, + prompt_messages=prompt_messages, + model_parameters=model_parameters, + tools=tools, + stop=stop, + stream=stream, + user=user, + ) + + def _generate( + self, + model: str, + credentials: dict, + prompt_messages: list[PromptMessage], + model_parameters: dict, + tools: list[PromptMessageTool] | None = None, + stop: list[str] | None = None, + stream: bool = True, + user: str | None = None, + ) -> LLMResult | Generator: + """ + Invoke large language model + + :param model: model name + :param credentials: credentials kwargs + :param prompt_messages: prompt messages + :param model_parameters: model parameters + :param stop: stop words + :param stream: is stream response + :param user: unique user id + :return: full response or stream response chunk generator result + """ + kwargs = self.to_kwargs(credentials) + # init model client + client = OpenAI(**kwargs) + + extra_model_kwargs = {} + if stop: + extra_model_kwargs["stop"] = stop + + if user: + extra_model_kwargs["user"] = user + + result = client.chat.completions.create( + messages=[self._convert_prompt_message_to_dict(m) for m in prompt_messages], + model=model, + stream=stream, + **model_parameters, + **extra_model_kwargs, + ) + + if stream: + return self._handle_chat_generate_stream_response( + model=model, credentials=credentials, response=result, tools=tools, prompt_messages=prompt_messages + ) + + return self._handle_chat_generate_response( + model=model, credentials=credentials, response=result, tools=tools, prompt_messages=prompt_messages + ) + + def _convert_prompt_message_to_dict(self, message: PromptMessage) -> dict: + """ + Convert PromptMessage to dict for OpenAI Compatibility API + """ + if isinstance(message, UserPromptMessage): + message = cast(UserPromptMessage, message) + if isinstance(message.content, str): + message_dict = {"role": "user", "content": message.content} + else: + raise ValueError("User message content must be str") + elif isinstance(message, AssistantPromptMessage): + message = cast(AssistantPromptMessage, message) + message_dict = {"role": "assistant", "content": message.content} + if message.tool_calls and len(message.tool_calls) > 0: + message_dict["function_call"] = { + "name": message.tool_calls[0].function.name, + "arguments": message.tool_calls[0].function.arguments, + } + elif isinstance(message, SystemPromptMessage): + message = cast(SystemPromptMessage, message) + message_dict = {"role": "system", "content": message.content} + elif isinstance(message, ToolPromptMessage): + # check if last message is user message + message = cast(ToolPromptMessage, message) + message_dict = {"role": "function", "content": message.content} + else: + raise ValueError(f"Unknown message type {type(message)}") + + return message_dict + + def _extract_response_tool_calls( + self, response_function_calls: list[FunctionCall] + ) -> list[AssistantPromptMessage.ToolCall]: + """ + Extract tool calls from response + + :param response_tool_calls: response tool calls + :return: list of tool calls + """ + tool_calls = [] + if response_function_calls: + for response_tool_call in response_function_calls: + function = AssistantPromptMessage.ToolCall.ToolCallFunction( + name=response_tool_call.name, arguments=response_tool_call.arguments + ) + + tool_call = AssistantPromptMessage.ToolCall(id=0, type="function", function=function) + tool_calls.append(tool_call) + + return tool_calls + + def to_kwargs(self, credentials: dict) -> dict: + """ + Convert invoke kwargs to client kwargs + + :param stream: is stream response + :param model_name: model name + :param credentials: credentials dict + :param model_parameters: model parameters + :return: client kwargs + """ + client_kwargs = { + "timeout": Timeout(315.0, read=300.0, write=10.0, connect=5.0), + "api_key": credentials.get('api_key', 'ollama'), + "base_url": str(URL(credentials["base_url"]) / "v1"), + } + + return client_kwargs + + def _handle_chat_generate_stream_response( + self, + model: str, + credentials: dict, + response: Stream[ChatCompletionChunk], + prompt_messages: list[PromptMessage], + tools: Optional[list[PromptMessageTool]] = None, + ) -> Generator: + full_response = "" + + for chunk in response: + if len(chunk.choices) == 0: + continue + + delta = chunk.choices[0] + + if delta.finish_reason is None and (delta.delta.content is None or delta.delta.content == ""): + continue + + # check if there is a tool call in the response + function_calls = None + if delta.delta.function_call: + function_calls = [delta.delta.function_call] + + assistant_message_tool_calls = self._extract_response_tool_calls(function_calls or []) + + # transform assistant message to prompt message + assistant_prompt_message = AssistantPromptMessage( + content=delta.delta.content or "", tool_calls=assistant_message_tool_calls + ) + + if delta.finish_reason is not None: + # temp_assistant_prompt_message is used to calculate usage + temp_assistant_prompt_message = AssistantPromptMessage( + content=full_response, tool_calls=assistant_message_tool_calls + ) + + prompt_tokens = self._num_tokens_from_messages(messages=prompt_messages, tools=tools) + completion_tokens = self._num_tokens_from_messages(messages=[temp_assistant_prompt_message], tools=[]) + + usage = self._calc_response_usage( + model=model, + credentials=credentials, + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + ) + + yield LLMResultChunk( + model=model, + prompt_messages=prompt_messages, + system_fingerprint=chunk.system_fingerprint, + delta=LLMResultChunkDelta( + index=delta.index, + message=assistant_prompt_message, + finish_reason=delta.finish_reason, + usage=usage, + ), + ) + else: + yield LLMResultChunk( + model=model, + prompt_messages=prompt_messages, + system_fingerprint=chunk.system_fingerprint, + delta=LLMResultChunkDelta( + index=delta.index, + message=assistant_prompt_message, + ), + ) + + full_response += delta.delta.content + + def _handle_chat_generate_response( + self, + model: str, + credentials: dict, + response: ChatCompletion, + prompt_messages: list[PromptMessage], + tools: Optional[list[PromptMessageTool]] = None, + ) -> LLMResult: + """ + Handle llm chat response + + :param model: model name + :param credentials: credentials + :param response: response + :param prompt_messages: prompt messages + :param tools: tools for tool calling + :return: llm response + """ + if len(response.choices) == 0: + raise InvokeServerUnavailableError("Empty response") + assistant_message = response.choices[0].message + + # convert function call to tool call + function_calls = assistant_message.function_call + tool_calls = self._extract_response_tool_calls([function_calls] if function_calls else []) + + # transform assistant message to prompt message + assistant_prompt_message = AssistantPromptMessage(content=assistant_message.content, tool_calls=tool_calls) + + prompt_tokens = self._num_tokens_from_messages(messages=prompt_messages, tools=tools) + completion_tokens = self._num_tokens_from_messages(messages=[assistant_prompt_message], tools=tools) + + usage = self._calc_response_usage( + model=model, credentials=credentials, prompt_tokens=prompt_tokens, completion_tokens=completion_tokens + ) + + response = LLMResult( + model=model, + prompt_messages=prompt_messages, + system_fingerprint=response.system_fingerprint, + usage=usage, + message=assistant_prompt_message, + ) + + return response + + def _num_tokens_from_messages( + self, messages: list[PromptMessage], tools: Optional[list[PromptMessageTool]] = None + ) -> int: + """Calculate num tokens for chatglm2 and chatglm3 with GPT2 tokenizer. + + it's too complex to calculate num tokens for chatglm2 and chatglm3 with ChatGLM tokenizer, + As a temporary solution we use GPT2 tokenizer instead. + + """ + + def tokens(text: str): + return self._get_num_tokens_by_gpt2(text) + + tokens_per_message = 3 + tokens_per_name = 1 + num_tokens = 0 + messages_dict = [self._convert_prompt_message_to_dict(m) for m in messages] + for message in messages_dict: + num_tokens += tokens_per_message + for key, value in message.items(): + if isinstance(value, list): + text = "" + for item in value: + if isinstance(item, dict) and item["type"] == "text": + text += item["text"] + value = text + + if key == "function_call": + for t_key, t_value in value.items(): + num_tokens += tokens(t_key) + if t_key == "function": + for f_key, f_value in t_value.items(): + num_tokens += tokens(f_key) + num_tokens += tokens(f_value) + else: + num_tokens += tokens(t_key) + num_tokens += tokens(t_value) + else: + num_tokens += tokens(str(value)) + + if key == "name": + num_tokens += tokens_per_name + + # every reply is primed with assistant + num_tokens += 3 + + if tools: + num_tokens += self._num_tokens_for_tools(tools) + + return num_tokens + + def _num_tokens_for_tools(self, tools: list[PromptMessageTool]) -> int: + """ + Calculate num tokens for tool calling + + :param encoding: encoding + :param tools: tools for tool calling + :return: number of tokens + """ + + def tokens(text: str): + return self._get_num_tokens_by_gpt2(text) + + num_tokens = 0 + for tool in tools: + # calculate num tokens for function object + num_tokens += tokens("name") + num_tokens += tokens(tool.name) + num_tokens += tokens("description") + num_tokens += tokens(tool.description) + parameters = tool.parameters + num_tokens += tokens("parameters") + num_tokens += tokens("type") + num_tokens += tokens(parameters.get("type")) + if "properties" in parameters: + num_tokens += tokens("properties") + for key, value in parameters.get("properties").items(): + num_tokens += tokens(key) + for field_key, field_value in value.items(): + num_tokens += tokens(field_key) + if field_key == "enum": + for enum_field in field_value: + num_tokens += 3 + num_tokens += tokens(enum_field) + else: + num_tokens += tokens(field_key) + num_tokens += tokens(str(field_value)) + if "required" in parameters: + num_tokens += tokens("required") + for required_field in parameters["required"]: + num_tokens += 3 + num_tokens += tokens(required_field) + + return num_tokens + + @classmethod + def agent_type(cls, response): + if response.lower().endswith('observation:'): + return 'react' + if 'observation:' not in response.lower() and 'action input:' in response.lower(): + return 'toolbench' + return None + + def get_num_tokens( + self, + model: str, + credentials: dict, + prompt_messages: list[PromptMessage], + tools: Optional[list[PromptMessageTool]] = None, + ) -> int: + return self._num_tokens_from_messages(prompt_messages, tools) + + def validate_credentials(self, model: str, credentials: Mapping) -> None: + pass + + @property + def _invoke_error_mapping(self) -> dict[type[InvokeError], list[type[Exception]]]: + """ + Map model invoke error to unified error + The key is the error type thrown to the caller + The value is the error type thrown by the model, + which needs to be converted into a unified error type for the caller. + + :return: Invoke error mapping + """ + return { + InvokeAuthorizationError: [ + requests.exceptions.InvalidHeader, # Missing or Invalid API Key + ], + InvokeBadRequestError: [ + requests.exceptions.HTTPError, # Invalid Endpoint URL or model name + requests.exceptions.InvalidURL, # Misconfigured request or other API error + ], + InvokeRateLimitError: [ + requests.exceptions.RetryError # Too many requests sent in a short period of time + ], + InvokeServerUnavailableError: [ + requests.exceptions.ConnectionError, # Engine Overloaded + requests.exceptions.HTTPError, # Server Error + ], + InvokeConnectionError: [ + requests.exceptions.ConnectTimeout, # Timeout + requests.exceptions.ReadTimeout, # Timeout + ], + } + + def get_customizable_model_schema(self, model: str, credentials: dict) -> AIModelEntity | None: + rules = [ + ParameterRule( + name='temperature', type=ParameterType.FLOAT, + use_template='temperature', + label=I18nObject( + zh_Hans='温度', en_US='Temperature' + ) + ), + ParameterRule( + name='top_p', type=ParameterType.FLOAT, + use_template='top_p', + label=I18nObject( + zh_Hans='Top P', en_US='Top P' + ) + ), + ParameterRule( + name='max_tokens', type=ParameterType.INT, + use_template='max_tokens', + min=1, + default=512, + label=I18nObject( + zh_Hans='最大生成长度', en_US='Max Tokens' + ) + ) + ] + + entity = AIModelEntity( + model=model, + label=I18nObject( + en_US=model + ), + fetch_from=FetchFrom.CUSTOMIZABLE_MODEL, + model_type=ModelType.LLM, + model_properties={'mode': 'chat'}, + parameter_rules=rules + ) + + return entity diff --git a/api/core/model_runtime/model_providers/modelscope/modelscope.py b/api/core/model_runtime/model_providers/modelscope/modelscope.py new file mode 100644 index 00000000000000..51bc9d51a41615 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/modelscope.py @@ -0,0 +1,10 @@ +import logging + +from core.model_runtime.model_providers.__base.model_provider import ModelProvider + +logger = logging.getLogger(__name__) + + +class ModelScopeProvider(ModelProvider): + def validate_provider_credentials(self, credentials: dict) -> None: + pass diff --git a/api/core/model_runtime/model_providers/modelscope/modelscope.yaml b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml new file mode 100644 index 00000000000000..3551060192853e --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml @@ -0,0 +1,38 @@ +provider: modelscope +label: + en_US: ModelScope +icon_small: + en_US: icon_s_en.png +icon_large: + en_US: icon_l_en.png +supported_model_types: +- llm +configurate_methods: +- customizable-model +model_credential_schema: + model: + label: + en_US: Model Name + zh_Hans: 模型名称 + placeholder: + en_US: Enter your model name + zh_Hans: 输入模型名称 + credential_form_schemas: + - variable: base_url + label: + zh_Hans: 服务器URL + en_US: Server url + type: secret-input + required: true + placeholder: + zh_Hans: 在此输入ModelScope服务地址 + en_US: Enter the server address of your ModelScope service + - variable: api_key + label: + zh_Hans: API_Key + en_US: API_Key + type: secret-input + required: true + placeholder: + zh_Hans: 在此输入API_Key + en_US: Enter the API_Key of your ModelScope service \ No newline at end of file From 2015ae662e68f9a226b0863ca1715e8be1362dc6 Mon Sep 17 00:00:00 2001 From: "yuze.zyz" Date: Thu, 5 Dec 2024 21:23:45 +0800 Subject: [PATCH 2/7] remove useless limitations --- .../model_runtime/model_providers/modelscope/llm/llm.py | 8 -------- .../model_providers/modelscope/modelscope.yaml | 2 +- 2 files changed, 1 insertion(+), 9 deletions(-) diff --git a/api/core/model_runtime/model_providers/modelscope/llm/llm.py b/api/core/model_runtime/model_providers/modelscope/llm/llm.py index bcd53b0f2f2a4c..0179e06efb4b8a 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/llm.py +++ b/api/core/model_runtime/model_providers/modelscope/llm/llm.py @@ -426,14 +426,6 @@ def tokens(text: str): return num_tokens - @classmethod - def agent_type(cls, response): - if response.lower().endswith('observation:'): - return 'react' - if 'observation:' not in response.lower() and 'action input:' in response.lower(): - return 'toolbench' - return None - def get_num_tokens( self, model: str, diff --git a/api/core/model_runtime/model_providers/modelscope/modelscope.yaml b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml index 3551060192853e..73c5816e138f10 100644 --- a/api/core/model_runtime/model_providers/modelscope/modelscope.yaml +++ b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml @@ -32,7 +32,7 @@ model_credential_schema: zh_Hans: API_Key en_US: API_Key type: secret-input - required: true + required: false placeholder: zh_Hans: 在此输入API_Key en_US: Enter the API_Key of your ModelScope service \ No newline at end of file From ab2f93e483d76ad0e70ce22f0437c60d629a47a9 Mon Sep 17 00:00:00 2001 From: "yuze.zyz" Date: Thu, 5 Dec 2024 21:24:21 +0800 Subject: [PATCH 3/7] lint --- .../model_providers/modelscope/llm/llm.py | 37 +++++-------------- 1 file changed, 9 insertions(+), 28 deletions(-) diff --git a/api/core/model_runtime/model_providers/modelscope/llm/llm.py b/api/core/model_runtime/model_providers/modelscope/llm/llm.py index 0179e06efb4b8a..98b11ccc8a7a8a 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/llm.py +++ b/api/core/model_runtime/model_providers/modelscope/llm/llm.py @@ -1,34 +1,21 @@ -from collections.abc import Generator, Iterator -from typing import cast, List, Optional, Union, Mapping +from collections.abc import Generator, Mapping +from typing import Optional, Union, cast + import requests +from httpx import Timeout from openai import ( - APIConnectionError, - APITimeoutError, - AuthenticationError, - ConflictError, - InternalServerError, - NotFoundError, OpenAI, - PermissionDeniedError, - RateLimitError, - UnprocessableEntityError, Stream, ) -from httpx import Timeout -from yarl import URL -from openai.types.chat import ChatCompletion, ChatCompletionChunk, ChatCompletionMessageToolCall -from openai.types.chat.chat_completion_chunk import ChoiceDeltaFunctionCall, ChoiceDeltaToolCall +from openai.types.chat import ChatCompletion, ChatCompletionChunk from openai.types.chat.chat_completion_message import FunctionCall -from openai.types.completion import Completion +from yarl import URL from core.model_runtime.entities.common_entities import I18nObject -from core.model_runtime.entities.llm_entities import LLMMode, LLMResult, LLMResultChunk, LLMResultChunkDelta +from core.model_runtime.entities.llm_entities import LLMResult, LLMResultChunk, LLMResultChunkDelta from core.model_runtime.entities.message_entities import ( AssistantPromptMessage, - ImagePromptMessageContent, PromptMessage, - PromptMessageContent, - PromptMessageContentType, PromptMessageTool, SystemPromptMessage, ToolPromptMessage, @@ -36,10 +23,7 @@ ) from core.model_runtime.entities.model_entities import ( AIModelEntity, - DefaultParameterName, FetchFrom, - ModelFeature, - ModelPropertyKey, ModelType, ParameterRule, ParameterType, @@ -52,10 +36,7 @@ InvokeRateLimitError, InvokeServerUnavailableError, ) -from core.model_runtime.errors.validate import CredentialsValidateFailedError from core.model_runtime.model_providers.__base.large_language_model import LargeLanguageModel -from core.model_runtime.model_providers.openai.llm.llm import OpenAILargeLanguageModel -from core.model_runtime.utils import helper class ModelScopeLargeLanguageModel(LargeLanguageModel): @@ -66,8 +47,8 @@ def _invoke( credentials: dict, prompt_messages: list[PromptMessage], model_parameters: dict, - tools: Optional[List[PromptMessageTool]] = None, - stop: Optional[List[str]] = None, + tools: Optional[list[PromptMessageTool]] = None, + stop: Optional[list[str]] = None, stream: bool = True, user: Optional[str] = None, ) -> Union[LLMResult, Generator]: From 2412dc18dec092b490893b130b8e48f064580582 Mon Sep 17 00:00:00 2001 From: "yuze.zyz" Date: Thu, 5 Dec 2024 21:29:00 +0800 Subject: [PATCH 4/7] add note --- .../model_runtime/model_providers/modelscope/modelscope.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/api/core/model_runtime/model_providers/modelscope/modelscope.yaml b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml index 73c5816e138f10..505ba850944fc6 100644 --- a/api/core/model_runtime/model_providers/modelscope/modelscope.yaml +++ b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml @@ -25,8 +25,8 @@ model_credential_schema: type: secret-input required: true placeholder: - zh_Hans: 在此输入ModelScope服务地址 - en_US: Enter the server address of your ModelScope service + zh_Hans: 支持SwingDeploy或者swift deploy两种服务 + en_US: Support SwingDeploy or swift deploy services - variable: api_key label: zh_Hans: API_Key From 6f2a8af3a9e408a85a020ae7b02e3e0b2d7c3634 Mon Sep 17 00:00:00 2001 From: "yuze.zyz" Date: Sat, 21 Dec 2024 17:30:40 +0800 Subject: [PATCH 5/7] improve user interface (cherry picked from commit ada33fa33bae0867d85a3dcbb26b3219a1368d5a) --- .../modelscope/llm/_position.yaml | 9 +++ .../llm/llama-3.3-70B-instruct.yaml | 77 +++++++++++++++++++ .../modelscope/llm/qwen2.5-14B-instruct.yaml | 76 ++++++++++++++++++ .../modelscope/llm/qwen2.5-32B-instruct.yaml | 76 ++++++++++++++++++ .../modelscope/llm/qwen2.5-72B-instruct.yaml | 76 ++++++++++++++++++ .../modelscope/llm/qwen2.5-7B-instruct.yaml | 76 ++++++++++++++++++ .../llm/qwen2.5-coder-14B-instruct.yaml | 76 ++++++++++++++++++ .../llm/qwen2.5-coder-32B-instruct.yaml | 76 ++++++++++++++++++ .../llm/qwen2.5-coder-7B-instruct.yaml | 76 ++++++++++++++++++ .../modelscope/llm/qwq-32B-preview.yaml | 76 ++++++++++++++++++ .../modelscope/modelscope.yaml | 29 ++++++- 11 files changed, 721 insertions(+), 2 deletions(-) create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/_position.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml create mode 100644 api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml diff --git a/api/core/model_runtime/model_providers/modelscope/llm/_position.yaml b/api/core/model_runtime/model_providers/modelscope/llm/_position.yaml new file mode 100644 index 00000000000000..880f9d8b60c2c0 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/_position.yaml @@ -0,0 +1,9 @@ +- qwq-32B-preview +- qwen2.5-7B-instruct +- qwen2.5-14B-instruct +- qwen2.5-32B-instruct +- qwen2.5-72B-instruct +- llama-3.3-70B-instruct +- qwen2.5-coder-7B-instruct +- qwen2.5-coder-14B-instruct +- qwen2.5-coder-32B-instruct diff --git a/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml new file mode 100644 index 00000000000000..0b6b7e6f281926 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml @@ -0,0 +1,77 @@ +# for more details, please refer to https://help.aliyun.com/zh/model-studio/getting-started/models +model: LLM-Research/Llama-3.3-70B-Instruct +label: + en_US: LLM-Research/Llama-3.3-70B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml new file mode 100644 index 00000000000000..cd29917baaa3f8 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml @@ -0,0 +1,76 @@ +model: Qwen/Qwen2.5-14B-Instruct +label: + en_US: Qwen/Qwen2.5-14B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml new file mode 100644 index 00000000000000..e3a598a2428b01 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml @@ -0,0 +1,76 @@ +model: Qwen/Qwen2.5-32B-Instruct +label: + en_US: Qwen/Qwen2.5-32B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml new file mode 100644 index 00000000000000..cec8768e8012c7 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml @@ -0,0 +1,76 @@ +model: Qwen/Qwen2.5-72B-Instruct +label: + en_US: Qwen/Qwen2.5-72B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml new file mode 100644 index 00000000000000..004d6472586d45 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml @@ -0,0 +1,76 @@ +model: Qwen/Qwen2.5-7B-Instruct +label: + en_US: Qwen/Qwen2.5-7B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml new file mode 100644 index 00000000000000..1063655662ad95 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml @@ -0,0 +1,76 @@ +model: Qwen/Qwen2.5-Coder-14B-Instruct +label: + en_US: Qwen/Qwen2.5-Coder-14B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml new file mode 100644 index 00000000000000..48ed6b8ca9f74e --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml @@ -0,0 +1,76 @@ +model: Qwen/Qwen2.5-Coder-32B-Instruct +label: + en_US: Qwen/Qwen2.5-Coder-32B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml new file mode 100644 index 00000000000000..c43b7d95823712 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml @@ -0,0 +1,76 @@ +model: Qwen/Qwen2.5-Coder-7B-Instruct +label: + en_US: Qwen/Qwen2.5-Coder-7B-Instruct +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml new file mode 100644 index 00000000000000..d50eeaf4933a63 --- /dev/null +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml @@ -0,0 +1,76 @@ +model: Qwen/QwQ-32B-Preview +label: + en_US: Qwen/QwQ-32B-Preview +model_type: llm +features: + - agent-thought + - tool-call + - stream-tool-call +model_properties: + mode: chat + context_size: 32768 +parameter_rules: + - name: temperature + use_template: temperature + type: float + default: 0.3 + min: 0.0 + max: 2.0 + help: + zh_Hans: 用于控制随机性和多样性的程度。具体来说,temperature值控制了生成文本时对每个候选词的概率分布进行平滑的程度。较高的temperature值会降低概率分布的峰值,使得更多的低概率词被选择,生成结果更加多样化;而较低的temperature值则会增强概率分布的峰值,使得高概率词更容易被选择,生成结果更加确定。 + en_US: Used to control the degree of randomness and diversity. Specifically, the temperature value controls the degree to which the probability distribution of each candidate word is smoothed when generating text. A higher temperature value will reduce the peak value of the probability distribution, allowing more low-probability words to be selected, and the generated results will be more diverse; while a lower temperature value will enhance the peak value of the probability distribution, making it easier for high-probability words to be selected. , the generated results are more certain. + - name: max_tokens + use_template: max_tokens + type: int + default: 8192 + min: 1 + max: 8192 + help: + zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 + en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. + - name: top_p + use_template: top_p + type: float + default: 0.8 + min: 0.1 + max: 0.9 + help: + zh_Hans: 生成过程中核采样方法概率阈值,例如,取值为0.8时,仅保留概率加起来大于等于0.8的最可能token的最小集合作为候选集。取值范围为(0,1.0),取值越大,生成的随机性越高;取值越低,生成的确定性越高。 + en_US: The probability threshold of the kernel sampling method during the generation process. For example, when the value is 0.8, only the smallest set of the most likely tokens with a sum of probabilities greater than or equal to 0.8 is retained as the candidate set. The value range is (0,1.0). The larger the value, the higher the randomness generated; the lower the value, the higher the certainty generated. + - name: top_k + type: int + min: 0 + max: 99 + label: + zh_Hans: 取样数量 + en_US: Top k + help: + zh_Hans: 生成时,采样候选集的大小。例如,取值为50时,仅将单次生成中得分最高的50个token组成随机采样的候选集。取值越大,生成的随机性越高;取值越小,生成的确定性越高。 + en_US: The size of the sample candidate set when generated. For example, when the value is 50, only the 50 highest-scoring tokens in a single generation form a randomly sampled candidate set. The larger the value, the higher the randomness generated; the smaller the value, the higher the certainty generated. + - name: seed + required: false + type: int + default: 1234 + label: + zh_Hans: 随机种子 + en_US: Random seed + help: + zh_Hans: 生成时使用的随机数种子,用户控制模型生成内容的随机性。支持无符号64位整数,默认值为 1234。在使用seed时,模型将尽可能生成相同或相似的结果,但目前不保证每次生成的结果完全相同。 + en_US: The random number seed used when generating, the user controls the randomness of the content generated by the model. Supports unsigned 64-bit integers, default value is 1234. When using seed, the model will try its best to generate the same or similar results, but there is currently no guarantee that the results will be exactly the same every time. + - name: repetition_penalty + required: false + type: float + default: 1.1 + label: + zh_Hans: 重复惩罚 + en_US: Repetition penalty + help: + zh_Hans: 用于控制模型生成时的重复度。提高repetition_penalty时可以降低模型生成的重复度。1.0表示不做惩罚。 + en_US: Used to control the repeatability when generating models. Increasing repetition_penalty can reduce the duplication of model generation. 1.0 means no punishment. + - name: response_format + use_template: response_format +pricing: + input: '0.000' + output: '0.000' + unit: '0.000' + currency: RMB diff --git a/api/core/model_runtime/model_providers/modelscope/modelscope.yaml b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml index 505ba850944fc6..01da61c9410fc1 100644 --- a/api/core/model_runtime/model_providers/modelscope/modelscope.yaml +++ b/api/core/model_runtime/model_providers/modelscope/modelscope.yaml @@ -5,10 +5,35 @@ icon_small: en_US: icon_s_en.png icon_large: en_US: icon_l_en.png +help: + title: + en_US: Get your API key from ModelScope + zh_Hans: 从魔搭社区获取API key + url: + en_US: https://www.modelscope.cn/docs/model-service/API-Inference/intro supported_model_types: - llm configurate_methods: +- predefined-model - customizable-model +provider_credential_schema: + credential_form_schemas: + - variable: base_url + label: + zh_Hans: 服务器URL + en_US: Server url + type: secret-input + required: true + default: https://api-inference.modelscope.cn + - variable: api_key + label: + zh_Hans: API_Key + en_US: API_Key + type: secret-input + required: false + placeholder: + zh_Hans: 在此输入API_Key + en_US: Enter the API_Key of your ModelScope service model_credential_schema: model: label: @@ -25,8 +50,8 @@ model_credential_schema: type: secret-input required: true placeholder: - zh_Hans: 支持SwingDeploy或者swift deploy两种服务 - en_US: Support SwingDeploy or swift deploy services + zh_Hans: ModelScope服务地址(API Inference/Swift/SwingDeploy) + en_US: ModelScope service url(API Inference/Swift/SwingDeploy) - variable: api_key label: zh_Hans: API_Key From a46e9de205d0a8fbbbc8b6fcf1983ab502692017 Mon Sep 17 00:00:00 2001 From: "yuze.zyz" Date: Sat, 21 Dec 2024 17:34:38 +0800 Subject: [PATCH 6/7] improve modelscope (cherry picked from commit 0e3f37d2a58a10b690c303d99bb754ad98add829) --- .../modelscope/llm/llama-3.3-70B-instruct.yaml | 4 ++-- .../model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml | 2 +- .../model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml | 2 +- .../model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml | 2 +- .../model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml | 2 +- .../modelscope/llm/qwen2.5-coder-14B-instruct.yaml | 2 +- .../modelscope/llm/qwen2.5-coder-32B-instruct.yaml | 2 +- .../modelscope/llm/qwen2.5-coder-7B-instruct.yaml | 2 +- .../model_providers/modelscope/llm/qwq-32B-preview.yaml | 2 +- 9 files changed, 10 insertions(+), 10 deletions(-) diff --git a/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml index 0b6b7e6f281926..c058f2d69895b7 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml @@ -9,7 +9,7 @@ features: - stream-tool-call model_properties: mode: chat - context_size: 32768 + context_size: 131072 parameter_rules: - name: temperature use_template: temperature @@ -25,7 +25,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 100000 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml index cd29917baaa3f8..a4d268325fa77c 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml index e3a598a2428b01..df01050b4c9f63 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml index cec8768e8012c7..0ad07529b8fde3 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml index 004d6472586d45..9d021cd9a9d0a4 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml index 1063655662ad95..c8e65dc76e8493 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml index 48ed6b8ca9f74e..a3871d57d164aa 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml index c43b7d95823712..2faf38973b5993 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml index d50eeaf4933a63..64d5e534b9460f 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml @@ -24,7 +24,7 @@ parameter_rules: type: int default: 8192 min: 1 - max: 8192 + max: 16384 help: zh_Hans: 用于指定模型在生成内容时token的最大数量,它定义了生成的上限,但不保证每次都会生成到这个数量。 en_US: It is used to specify the maximum number of tokens when the model generates content. It defines the upper limit of generation, but does not guarantee that this number will be generated every time. From 9d87ef9528af9e23173c288393cc1c4071e16782 Mon Sep 17 00:00:00 2001 From: "yuze.zyz" Date: Mon, 23 Dec 2024 17:19:44 +0800 Subject: [PATCH 7/7] fix --- .../modelscope/llm/llama-3.3-70B-instruct.yaml | 4 ---- .../model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml | 4 ---- .../model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml | 4 ---- .../model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml | 4 ---- .../model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml | 4 ---- .../modelscope/llm/qwen2.5-coder-14B-instruct.yaml | 4 ---- .../modelscope/llm/qwen2.5-coder-32B-instruct.yaml | 4 ---- .../modelscope/llm/qwen2.5-coder-7B-instruct.yaml | 4 ---- .../model_providers/modelscope/llm/qwq-32B-preview.yaml | 4 ---- 9 files changed, 36 deletions(-) diff --git a/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml index c058f2d69895b7..0acbb62a0f5a0c 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/llama-3.3-70B-instruct.yaml @@ -3,10 +3,6 @@ model: LLM-Research/Llama-3.3-70B-Instruct label: en_US: LLM-Research/Llama-3.3-70B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 131072 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml index a4d268325fa77c..e063f903649238 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-14B-instruct.yaml @@ -2,10 +2,6 @@ model: Qwen/Qwen2.5-14B-Instruct label: en_US: Qwen/Qwen2.5-14B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml index df01050b4c9f63..cecd85661dad11 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-32B-instruct.yaml @@ -2,10 +2,6 @@ model: Qwen/Qwen2.5-32B-Instruct label: en_US: Qwen/Qwen2.5-32B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml index 0ad07529b8fde3..b3131d63f0b2ea 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-72B-instruct.yaml @@ -2,10 +2,6 @@ model: Qwen/Qwen2.5-72B-Instruct label: en_US: Qwen/Qwen2.5-72B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml index 9d021cd9a9d0a4..a96013f6464b06 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-7B-instruct.yaml @@ -2,10 +2,6 @@ model: Qwen/Qwen2.5-7B-Instruct label: en_US: Qwen/Qwen2.5-7B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml index c8e65dc76e8493..e038d325117f9d 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-14B-instruct.yaml @@ -2,10 +2,6 @@ model: Qwen/Qwen2.5-Coder-14B-Instruct label: en_US: Qwen/Qwen2.5-Coder-14B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml index a3871d57d164aa..c3f018268a606b 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-32B-instruct.yaml @@ -2,10 +2,6 @@ model: Qwen/Qwen2.5-Coder-32B-Instruct label: en_US: Qwen/Qwen2.5-Coder-32B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml index 2faf38973b5993..babc7cb1801dc1 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwen2.5-coder-7B-instruct.yaml @@ -2,10 +2,6 @@ model: Qwen/Qwen2.5-Coder-7B-Instruct label: en_US: Qwen/Qwen2.5-Coder-7B-Instruct model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768 diff --git a/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml b/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml index 64d5e534b9460f..4d6f6f6daf4a3a 100644 --- a/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml +++ b/api/core/model_runtime/model_providers/modelscope/llm/qwq-32B-preview.yaml @@ -2,10 +2,6 @@ model: Qwen/QwQ-32B-Preview label: en_US: Qwen/QwQ-32B-Preview model_type: llm -features: - - agent-thought - - tool-call - - stream-tool-call model_properties: mode: chat context_size: 32768