From 0bf4a18b22e8bb8718d95294e9f7f45c0d4270a4 Mon Sep 17 00:00:00 2001 From: Pascal Date: Thu, 11 Jun 2026 21:45:13 +0200 Subject: [PATCH] codec: add pre-encoded voice reference (--ref-spk / --ref-rvq) qwen-codec --talker extracts the speaker embedding (.spk, raw f32) and the ICL codes (.rvq) in one pass, encode truncated to the hop boundary conforming to the --ref-wav path. qwen-tts loads them via --ref-spk / --ref-rvq and skips the speaker encoder and codec encode on every synthesis: TTFA 205 ms -> 89 ms. Extends qt_tts_params with ABI v2 latent fields, adds qt_num_codebooks(), ships freeman.spk + freeman.rvq and switches clone scripts to the latent path. Output is bit-identical to the raw path at fixed seed. --- README.md | 16 ++++ examples/clone.cmd | 3 +- examples/clone.sh | 3 +- examples/freeman.rvq | Bin 0 -> 4730 bytes examples/freeman.spk | Bin 0 -> 8192 bytes src/pipeline-tts.cpp | 80 ++++++++++++---- src/qwen.cpp | 28 ++++-- src/qwen.h | 21 ++++- src/rvq-file.h | 111 ++++++++++++++++++++++ tools/qwen-codec.cpp | 215 ++++++++++++++++++++----------------------- tools/qwen-tts.cpp | 66 +++++++++++++ 11 files changed, 399 insertions(+), 144 deletions(-) create mode 100644 examples/freeman.rvq create mode 100644 examples/freeman.spk create mode 100644 src/rvq-file.h diff --git a/README.md b/README.md index bf9ecc8..6ef0206 100644 --- a/README.md +++ b/README.md @@ -82,6 +82,22 @@ Voice cloning (`clone.sh`, Base, reference WAV plus its transcript) : --lang English -o out.wav < prompt.txt ``` +Pre-encoded reference (`clone.sh`): `qwen-codec --talker` encodes a reference +WAV into two compact latents in one pass, the `.spk` speaker embedding and +the `.rvq` ICL codes, bit-identical to what the `--ref-wav` path computes +internally. Passing them via `--ref-spk` / `--ref-rvq` skips the speaker +encoder and the codec encode on every synthesis: + +``` +build/qwen-codec --model models/qwen-tokenizer-12hz-Q8_0.gguf \ + --talker models/qwen-talker-1.7b-base-Q8_0.gguf -i ref.wav +build/qwen-tts \ + --model models/qwen-talker-1.7b-base-Q8_0.gguf \ + --codec models/qwen-tokenizer-12hz-Q8_0.gguf \ + --ref-spk ref.spk --ref-rvq ref.rvq --ref-text ref.txt \ + --lang English -o out.wav < prompt.txt +``` + Named speaker (`customvoice.sh`, CustomVoice) : ``` diff --git a/examples/clone.cmd b/examples/clone.cmd index 57b591e..57b9754 100644 --- a/examples/clone.cmd +++ b/examples/clone.cmd @@ -5,7 +5,8 @@ set PATH=%~dp0..\build\Release;%PATH% qwen-tts.exe ^ --model ..\models\qwen-talker-1.7b-base-Q8_0.gguf ^ --codec ..\models\qwen-tokenizer-12hz-Q8_0.gguf ^ - --ref-wav freeman.wav ^ + --ref-spk freeman.spk ^ + --ref-rvq freeman.rvq ^ --ref-text freeman.txt ^ --lang English ^ -o clone.wav < prompt.txt diff --git a/examples/clone.sh b/examples/clone.sh index a98cf1c..2b934b2 100755 --- a/examples/clone.sh +++ b/examples/clone.sh @@ -5,7 +5,8 @@ set -eu ../build/qwen-tts \ --model ../models/qwen-talker-1.7b-base-Q8_0.gguf \ --codec ../models/qwen-tokenizer-12hz-Q8_0.gguf \ - --ref-wav freeman.wav \ + --ref-spk freeman.spk \ + --ref-rvq freeman.rvq \ --ref-text freeman.txt \ --lang English \ -o clone.wav < prompt.txt diff --git a/examples/freeman.rvq b/examples/freeman.rvq new file mode 100644 index 0000000000000000000000000000000000000000..06d122f7114abe7e1147b2e6042741745be0c81f GIT binary patch literal 4730 zcmV-=5{2!>)PH~EJNRBc?s4d?TkIe55E0k1>j%WX5^!9q(3$w@WuZ%c6;6<+q&dy# z?46<%72HYf_h;3ucTpU~N z_{Mj=OMUZFm%=r!FZIoDe)F5(|K>Np`OR;B^PAr+!we+eFs<0_-?^JhY~&fnM663) zL%f{(ZRvkqOU_p!7ICdb*L~|tj_j+w-u13`sY`uRa^L)>73q zdkv)Bpn-&6LwLV4_hj#-E#V;`&c<5o9$a9Lo_K@`ht~k)s71g7GV{NSaGUvlXL)W` zN@2n?zd`OHG2sE*qIeM1d?d=75X_NIkAogI24|I(3E`D|J7z=<2fDWSd z+LVz?S~?t#j-YcgOHTWlfc_T&6Ks5X@tdPfB4Qp#_(N_@m0IdIxgX{`oh9d9-LW)I z(FHAQBC7&UfT5>&u3S*HEUMm>O>L06G3sFD0#C_9C@c4zC#}1#G3n@ki zpmvK`7fGIunz+znss)a6&JQi^a+9~{f-T~#?tBZu`UJd*u%gQB+-7(2K}3xGrv!?f!y386({KHzo^H zsF0!Ixw7&U(rwcPaG{@%VOXtj^@=*xO%WxRLiaRh_$P?=709STE-GK0n~eq4(;^IZ7+zDntO?1ZjQ6kCab<-9^Q z&9b&`Z^g${E((PF|LF0J$R%!7={uPdNbuGCKV6Uajq>zojFwri<`)(rPdzu!ynG>s zT(Lo__dF`|`2C7B$Fj(d_3+f=##3#S{r8*~G)p*~R?2@%|OO z&T|FU-_4dXDn;`^JY2G&D)NTfUnYACd_tX#dt5Kho~b|L(Wl)=P7IJ}{|7-S9vADM z0z*6JL-ojr|LlH%dDbxH0L8MvvPL_sbk)hk-98hr?g#sLm5-Q_3I(o?J`lx2O>&FE zVr5ZCQUk-$IAW!XV2uNmjSR?Fm);EXNyMAjs?da_jI%;@eFZGP9Re!dk$IWrk_=D- zil^7_=?^TM%>}b>+8~nMaYX6zTW3-0;Xv&OM^IEYbADbiOH)yS#~0ezXWPy$M(jt( zs)zS+`lDMqd|?Fz+E`h-D`V`N>>66E2DB*g8_Pwz)Bcs}$dYlQtLjbDig2OL!YOMD zcZvO+v|YNU5zC11*8OXTzl>)0qgVoF$Jai6)h5lPEQp^@Vk2~|3Wa58S%2R> zJSAW~8NhZ6u0f}~)}-SRVX=h(wqw1$I(~_OzWCWBXZ~rG*>!M`(wq9Fke~?NSAM%b z?og{c^doo-;xxq%aO9cZqbWOcQz9`)6 z1gkPcU(B$B<+2kWPz)O}aNmF*)^!NC#m@8*DlOU?2QAW^QHjrcWwpB&@dttCVkPvV zR%m*v%rNF+#9rqs_7{4U)C@~1-{3r2)D4x*+|3A1!XrnkN?{2CMX8bn9{5D&ux+Dr zWVV3W0Kbet=NY}dMf~V(%lXDt6n)0HP_T|qk@d2dJ+z6M?9!agwyMOA?;?|n!z;lf zP-H5JaVTpltbH6mu6@5!5@#x4CP zTVv*b9&jpP6aIHav3^nqLFJDt{>BpiwY_RzF~vto@y_no_zv zS)A3L+FF&43zrn;>e71eq=MpN=yLv;R1X8=L+Y%%F7_bW+d1U9*!aut`tmBk1l50yTmCXUl?{8|74cij;xFOJ z7={;ijPl{`kS`dDd)KT@^6zb2yAc+GgFLd4Drpw*WQpzK_lpavC4jqokN(Sx1yU4n zKubWt8}1R(;^*sWSOPJ|+LoScZRu)al~+Q7476ONxm_6tv*~#x*>29mR<-OCFkN|= z>&b(3wOWFYi6rv+xgxUqK)rtv-_rX=2|9wVdHjr3DrQ!TTvlR!2r{i;k!eN$g}`hU z*6{jYT=4N=sn>42W}?`wNYSy-$P=UarL``Fj+v z%dr7Z(@?&bB>Y;B=O7$lQ3YLV9{He@gquj8%~;88*C!LzVl~LS_No)VI$nd`X$2!^iz-&(b^IPY_b4>xyy)--R4}IoF`b)*KCGq8IGONV0C~ zYuB;0i(j1fzFJ_ML%#3h1lEI~a;P~49qA53RCeXK4kt%h6hO#Qn(^gDM$*s^v|lf3>xJ>{!`lk?zZ#q#H%}d3fh+XKcv8LVof7XZN(_G-0nnDY z5RtoAx2Oi;y11TfY#Onhf z6Jr1eUVYP_HSUwH!$KLg?2;LTZ5?ddJEDxS;9AYEgWTgb5Ipwbjd)6S(sw?m0-H-I z(xaeN;4Uam-8xJ&p2bie(D#85-@q!wkI$y$L{9+MoKVkbtf=;9CB#r#&)d=$BXy-$ zC22&UHM`Ko5X9mI_|Q#V09CaVSf!+K&BwlGmyCk{fdJMR&2X!ijRla%$^e5xT)_WT zYjU?uzF&HfHZ84n_s`|1X4hm^a0p9vKPN07dd5y~=)kAaJ0!Ob0w?9cC>G0pb3Z)f z*bV6|ne%L>hM{ZI$gsazNe>J~0)7QJ!ABc%25Ql4F`yW(aj2sUbaXq}2d5O)M=~l= zje~aHOic)U_Iy|1H=Da85!bd^3=mS52Dh-Ysfu8>bfK=pv5G1rY^f3r{#bIyL^ck* zUWcw@ZFkagS0b6K6PYU~#q_`+cDnegB*}fl`H{qXKb_)NS$lSF6k&q-=ol;&)gW(8 zVmCNw4fXp^Pkh1cwUPA7%BMYWRttHyN&DL!p7tO_B1nIk!x$sPFX1n}F~d-Hp4`*M3!LC`$}4 z(nHI)xu!;`U3wKTM&?DPgF(c`C`1Q13WGC9));r~Qq`{J67ktBHv`S!9y={-W0#o{ z2L%N=QxVxu#YE6cz(;^aMuJc>l37h;U-iAgKrurJE-7Hqr3BS?xn_53QadEd)+OTL zu+T>Y3FDOaC031vYY_j4@vQ)H0y)g00_8cc8~uh?b1DN6R$)^4A}MB3mMo}W{Pe9J zrqLhTNC9`ow!#d{&&sGs;$1L*fU#h#Z5Rc+ixT~Eu#P}Mp7-n_YCOL`ai2Av0|v%X zBMXG`&h?tmCn-|Y_8EnR1bocN4)ki_@UH57{hC;zHE|rO-2bWk6q|9w+$j*gzqhgt zS}WXhzKf7|Sfm0ay{uF{f6Zmy!k}I`>4Q~4j48hanp2i5e=EaJhYedVE%+`rToVgPcpqBM zJoL-oX?G%fw55lXvKkLIK3m(9%Ye;DmXtCQj(x};SBv(sSor~(WU*nU#I%$QJs0-z zUd#=Zd9WjR`6!aG@D1bCw+iaU!Ojy<$SMAz_9RdH36*i)e}8PI78Ms#ty5M$(%wR0 zIsu*9z!H5yocW0i4FENRROTIA=zHMVL?f5mm1yRPTUP>I+X zjOK$Ig9IgNx4#@Hm&dYIDzeNUX?m6hvGEFH-e-t_Hx?vTrt{`)f$**=53RGOrkR5o zC_gNv@xRe4^@FtAj9(^~DZ$BrvUr2{cTkZO{_R9Cdmj*4iA)#Qhd0HhGXZsG!BN(D z&TqbZri7K=WI7*JVMX4hW_12^?F|=|}=1jazl!Hu~8Y&M4ppoDzG@*Qniz zEtBTz-h)$EynII!vV4wQS+&nTmq0H(_3B_~(_2Wa5G+h_5mBIh3sbkFw}T6t_Jq-t zQOeiKx+`iM4i7E_`jI;>L^-8YY|#48+dEpGVUP}O&&OY7C;rF0ps^tGw0k_QPY@n5uMS&PKWLRPoW(KL z#a+yf772%%FDC>ex-t_9tA*8R literal 0 HcmV?d00001 diff --git a/examples/freeman.spk b/examples/freeman.spk new file mode 100644 index 0000000000000000000000000000000000000000..e57f209ec117f98e2504225cd0f6d681402c4d20 GIT binary patch literal 8192 zcmWNWjYEuS8^uQ-Oi5CqXlq1C!bsXQ_w_WAL_Sg>Nk$SIGPX8peT*cPWF$!?Bb7>+ zl2mhFPeW1YrAV73gQOBdY;5%QKV0XW-*wL1op-_Z@+~Tr&Znx}=K`sVAtC$6!57YH z`0#eS(kJ#1bY!*Q%=l~Mf3_MJA7scT$kajTE5JFx;wDi7s7+TgW2RiAOP?Bp_ryVz zE1l4x>~9ngk0B+aGT~VH2sV6{2fNGOfpPwjfqF}#1f9Jx0`8|6;>8}tc{gUVJ)O?L zS(~iL)_+54i$kGI#{le}+vDCzJoMN&heoV~LmH680OSkt_T=?b!D3pnPE#?;t0S{9axoz54@FX0Nz4xUpvpdrMvBcbJALvS<;2YJ*05kBq{BrK>SJ(ujj z@9;Ec_Jb;ju#r=>-#4<}S2uuPYX>DuIOM^JyZDZqjOxeo;on`$FpTwqs-`9=`P@Xe zI&_e_hwfOW>>}K=4>0XR667|Xg2>rOuAhC36Z{u2p*Dwc`#cE=$=%3sj5Ubtq(1qk zql5)2DJw76g5RB5^Py*9E{*IAfu0O| zG`f9GU|BE*s@;E5U+{;nh&!;~{~72xxUhEoHPE1*fW1G;KsE9#bT*a|Uf?16WBOCd zyStSTWi@%YWIh|5It9*ave+FILEF-taOwFfaG(7IC9C5F!azHc>~Sl{Rmxfvy_>7ZOy6?4=JIGZ(ozRQ;;T?)c~EiR zCkP*mih`Qq`50Ug1AOyQ7&$Kmc_SZ_BZm55^T&H~{DllT&&;T^mOm=jqQGUWh(`Z( zM9HRI_~)ub9Qge%?X4JrR(wLuwSU8`vvb*3QO8)(FD_`kXodrf8tXCc0@*m&2-=UU z(P5>6?hPoX{JHCB)ZBledQ>zD&0A>WuEmfYFr*R76`_!>cP3EoVrMG*zu-o28yTy~y zLw^meTfUolzJ3NL;QHw`K=q!N%xQ*@_;g-jIPNs2R-!$vmAgPm&$I(Xd#agiyhadZKLZ|7L{N;~+;8bFfW zuaK-cNlPB3q5aAoD1ieN%5BF5^?6BTiv2Z^@BBb@@8~e*De9~^X$&RhqZxHsEN~p& zItHHp1qowwq1fCI^LqXf@ZU{k#oTLx?s*SEZ}y-f_S;=LOQRUWPh3OwQRCRkhO3zR z)DI($^02kzAd!voQpmPmEt=6~=UFORu0}cEXH!|%7xL<= zIh2_u(;N14aMT}HVE>jzNK;<}689g1g#bqglHPk*Id8lU@mY z1PSXDq%2sf;AaHW46Qv_GyM`uSB%6y5P2 zxV!%&{KdJr)=R=lHZ8-;XZ!I^xe@x5Y^Sxy#E=l+gvI?ir211P3Pa;*qsj?w=H#K| z@FnQ&^=Eu8zoiys`_Sm85nKu;Y-Oc3h(k5Wl7UvRy?z9?jakV`YDyIa#ukwILqe+4 zJE=KnGD||Y;Xp zza{zy*P=A@CS95|7dXdW3z~XIAm1mDaCBq>o~JIcKJI}^_9zK?`H2R9jRwhz3ku1` zczlri8)Ne*mlT+eKy$Ax&|)$Y>H>pc@9kD-{+mzkN!qbv{9@n*FQaBl@}XbX2ns&F zM*f(q7#6e|r;K?>IoliPktZCeU#rJ(+(5xEPsLf4_MoHHfaW(xfywPqD%n)zNZ$;A zXz?5rZ#DwC-AC%a)|8>)8ju&jN1IK@iCgs|`erbXv8~D^lC=52Rg^&1tuq)~Tm*TF zBC;asDI@HDX{!=zPfH^hboW8e3;D114g;=vR%RNJG?3P0-#64r*Isd5x+vu`YRS$Kl9Qy)aT zJ}5G5W2m8N6Lm0aCoYBW>GlB!I%5Wp36y+5!=4H@sHq2&9gXBy+&Ngf(i45YXW&PR z7~pvrpu|j@Hh=ae%@zy5eVqwpsU3%Q+JSfeaDWGN8Av=?$DWM=Xyu_pubUKN&}%j3 zuy7E)QJXFIGht_>%i!T(qj195Qz#j`9#@5DfKE{rCg^aPmZO;%v9VO3`d<&JJ-h&7 z;wG`GL*7K{caub283!d}GO1qqF*I8zLGHW+s<$qRiZ65%qf1Aqn5#tZZPVb6(ge09 zThWX8A4v}eVr|}##F8GV`%%3|B{JRqDjBnA-8FSH=Xrhr{8I=znp?_{V zgKm3fqUNiMK)x3OXI~N#Yb2q#)q%RIWfPU_6|!#nMO>=KB_4iX6s~GAdNt7k5{yry z^i-mt{^WYBPP>OPWvZez%MHr*X%XR(8WKBpBs;9|KJc!%Qtt(7OiVx#_>aO=mJJzh!VS`ru^9EJT>_es5v2l?!_8w%|XKwj)D za-sh}l*a`SI)e+9-m8$masmXdnFSnW)$W?1$x z6#73fls`>Q`*a8B@pa#E*BlK{{e1~dO7&^@pYa%{_YuGKIh*DP?)kHhGoI2b)xg`&c*0>9XHlGta6iR%^#q9R!=-RH#|G zo-N}_&^JPtMQc7}PE5hs!Ym@Rm`O5hCPVD$aC}5HoQ!T2((P6p6nZ!#=V)B{BRGh= zrE*lss}%JoGZcCQTw1a(4!OgXG%a2a&K^quRZKXMsI(ZP_D1CQOX1(B3cUNI9VIW1 zu{w-36u6B;{*^fbjeB3n;t5$$xM&a~e>tJAQ#{stsStLjGD`R5 zF-1{*0lNEB$lLQnA?G#F5YH)~YdHo*DLU*q&QR_QW$3H!bbOSv5E|N-!YkWzC>^;C z0-qHT)72>;)2M|X^$S5%aR_`imkK&3Y$1k`eAdad0)@@$^m9@Lcs`p8olg10=8^|4 zKbHxr@Gms)hm!VQx=!38f%GTVqUg|n;Jsocl>M*+dFEWgzn@BZlZ!}+5tmub%0Xff zLq3mLi@dH#Hc*npy4XA;A98Cjb^-$=s7yx(i+V=oGc)c zrk@q^qP^5*#t1aAKaU(vc)9Ql55k_s5Y;*tY_961)?NM7Rni4+m6gP>DHn6&cv^>KIc$dKaOY+dzKuh$5rxF&yt3&b+G4 zN5?gKn03?G>^z1MOiPjvPy0;!B4+?2CO5~ zD1X=qvNLZ4TdyP&Rr)j7a$60UpJUnPlXnxaAl=$DmTp@E`PE z`WzHc%?db2uxHVIOijsw-~aatYE8m`A2ODm>DmrT%}Ih^$0mX6yd&V}yByuFI$>Dd zGV{VieBRS0#eD-L?gtr@5{)6YRerXq!T$E$~nI)KKY0O^u zBa~@OaiF>@8$qaAt%z8?jEFm;F#6>Sa$w6cR82CY32+U6$ZXkP_x}M^aVe?&?*_Xr z$b{*R&xMd3-5A(W0GTVJQM7I%XwNgjJ5g=mXKsQHInyClxEhVrEYZj8v|uVNq5A8&)vbAxb(xgn$RX)(I_j;yR}T#25e&!faM02ASo15;fS4C%!IVW{96* z#JL@$=YauboI+$Q~f-DLaX4)8fP|42`*e{AA=0!WOWT7sT zI%rKQEIz^Z%MEyat_~>AyJ3a;Qr7*#PDs6Jh~j5nV7Ye*xa{~!^!)D-q4q|SF~Skx z(R8N2>l1njr+~IfhxScBORay?g3tuEhc+SDa{RMhuOv zbAqH83)p)eSobP@R`^wo3JrN68F)(yRcRt7D>29J~aBb_S!;XD5kpy+pPfe1X0`Ek^Q}D|wh~ z0zAt_G-H?>CeM9{ycd}Q?dn?4f7wBLr<6eIs)@|^!5I)Oa7X@)D(bcW5R+UQLtT2( zFcV_%a~p$cHW?`P3W9&u{0W8@Z>YhgE;ynpf{mlqvGiyir2Sq`>lb*C)j`o1zh(`} zUA_?I*b1uBe~nAIjnwl;Hi+ln6*Sw=WG$2{7?;@@c&pWnv247-W)6EmUj-dSc}pc0 z6c_`$un?B~3IVk{xuBJH4V4A93Q5jrh0&RJFy22Ov&$?oSQ-LG3EQ#f)>xu9UfjCg7CbDDfGJ+ zS021mL+M`uSTZ$)%Gk$r`D{7zB36-1dr_rI;Ya$<59*y!3rb6wt(O`No9CSYK#oF~AC|oG5*vd6wB1h!F)wv@X z*`5>h*@LQ~USzI94Cm<0%~w#BYeG5cxdOk@spQbEJox*6*YHWhJ@!_%D^#0bN2A1h zn0bBxT)0cgm9S1Iin)evnLkL$n<%`z!HBJY`+#~Jx(HLgYp@CL6LIE`8myhM8)Qiy zFd;e=!uH=5v=o&w*BZvNGb`VLWm!CO$2ZVUwRu!IFN*Y=t^?tzAun2PfX%y&=w6c~ z`onlCy8TQfBMK-9t4_&BoL5>@A16>c^!>CCJ{cx$p64z+cR z&->9Z%X5eotVgkz7Zl;@?kS8!|0o$f#L)?kX4uuiW%L#=!OcVWA!5|Avv>eI&hwBr zjEm9h_1U>*<5~axXV7ldN+!?#8J^(BQCXmnr0{RTt#j8PueqA!I>(^WHQUi@=|N&@ z-M}QmaA>UDh&E*djKj+U9C7s_^gf*iVHRUiY3`=5$#*3!4umZzI|3gYk3(k63y3bd zOJl;0v8sRcsm8SfC=A*x&^Tob!xOUEv%NRa>3>_GweTMF{i`5$yi_=G`yTSftcKLv z56Q3dr@^jzBP8s(2D}vy6+DTI@QNf?tPv=y*XJ9nXd_)yh5_nr9#nO z$JV|LQ%K7mkx0v*#PwxA{Mt1ER=x;e|MgsiW3>}uO|pbJs;kdj9hVLp)>^WWxuzRG*5+z2g|+icdsi^<{9LR*2zUO{f>VfC+o( zje+62N$p>uByU~{>1iy-jK{;N%f&Bv=(+~0dODZ(pD`yp-f)q7?-$ljY(lEx2vMS= zC}{~N@-rc%w&)G{JQ#|f?fzjr!a4}bwP9~+1Njza0#=vSVC<=3Xtit$75||})4j`> zzh-_zi&x{=tfnd~<^GQr4bKFLu>p;|S_=*z6N#i|FVUYU#0;G~$WJO|>rSs?txbhE zDo7preM^zYR0$mP(Yc$#BgggF_t6@9D^5V{`#^|o zIV9j2JXO^BvY=`hO{Z+phJjjJw)TP}Bv}lD6CtUv_em=HFE~be+j3Eu_zC-uKBp;% zZ&PKL0d4x|3-#hMqI8eOSP>6X_8!EF`B7AoAEMB2NJN_sS7&uUmsNZj(-f=sK z{=|nZ4TkJhy%f~=bPYrDJ22_L-@s?`P1^WX1m+9m_@{x23OOr~$$Uk%v)7?~`eU5$ zFM#9-D{MV$hF*DMrt@bK`W>4@YHxieWfL<yhlcOCV(caO$=+zVn; zK6y3kZwT9xM5V!Ju-3;1(~G4b8)l7bM}CH^?WLrArwn$CyvIgpzMumCON=`tn% zN%22aIlPGqn|l-!iaPMY;;ERD{svw0FJWM^11sCar+D!Vs)kl9t5w#Z^0F65f;DOB z#!;x7bb^TW9@7+qWN2~M0eSvLGCej1JzNqXE8!s4`#cqhhg}k=mQ7&)oL$5`b=AdE z&lwQ2whhJUVyZD?6s8=PV*6EXCb_wj)J;;N+rC6v{P{fX99BkKE39$yyBw4ZY@$Z< zSAfRqD-aZWn0QudLFyC>;Jp1rk_$Akw@r@GNu5-=EDBRYv%v7P6K1ZM$e7wzK)^y1 zw*7<|dK*Pjk6CNjtPwXb)%qX8(KScIM;gqhYu4DUkxpzGDRDZgi$0#4X_##@@$TS) z1~G*E4k`G!4tdU>VMKaPggPElT6_E+4W6QpXP-R*(Vgkwdu$e?NWTI8T2H|3`Xo&L zT8A+gc$mW(!}=P@fgjvQek(o5L^ce9x@#<2jk!b$+jgVS(ODs$qYY)#ve576DXf_A z3U_GMfb*7DaNThuE7`~=uDBPolh#qwOnV$QeIjG8*ba6QZS37Wj5cSvfZf{!eDL?p zp$+g2RC#b&UyCzn>VF0vyE`%U-Jy!Qm(A3}?=cEB0~GVM7BNjF8&IBIrHEV+1(Miz zjpqy_%1OTK|!VbNdNLqOQ17&uhAx>mr6tSC>oFrV>ITY)Nf zXNa%63Ualpv~I^vs^;vDon~Q}mY_hpQ-!1_;sq)2IRwA2D22#7Ux1T&5Ng6U1CNS; zlavpV`)@%bI}SO874!7ErZ7s&#R8+iGE95<73R}(;4`TXk{1bK$?JGRE%g|;<%?ik zaX$96eFw3xIrMgJA|6j?1GnTjiP#?wM-$Fr*vVE}VYnWp)4fRcq;DiQU^tuba5_8c zFb|9}8sPW7QfTYm3KI`4f`F3CAQD6qad;q;DH5=`&PADq8};8-gdLtbD4fc{`Zw?dO8x;LbZ-B{AXWZ^&0f%`~q4&0h ze0DjHdruvq15b{Th!Lg4^hq|z=U3CdTYr$<`L~dhv}@>>O(l+b2F#oBhf#jGjntkV z3rdk2le=#*T)uY>L+oNmtwTHd9qFWvN4>H6vj$X*NCAFm6Ac?>1j_VV;C<8|9}d%I Id70k$e|oHp6aWAK literal 0 HcmV?d00001 diff --git a/src/pipeline-tts.cpp b/src/pipeline-tts.cpp index e00ab42..4512004 100644 --- a/src/pipeline-tts.cpp +++ b/src/pipeline-tts.cpp @@ -376,14 +376,50 @@ qt_status pipeline_tts_synthesize(PipelineTTS * pt, const std::string speaker = params->speaker ? params->speaker : ""; const std::string ref_text = params->ref_text ? params->ref_text : ""; - // Voice clone mode A: if ref_audio_24k is given, run the speaker - // encoder on the pre-decoded mono buffer and feed the resulting - // embedding straight into the prompt builder. Mutually exclusive - // with --speaker. - const bool has_ref_audio = (params->ref_audio_24k != NULL) && (params->ref_n_samples > 0); + // ABI v2 latent reference fields. Callers compiled against ABI 1 + // never set them; the abi_version gate keeps their uninitialised + // tail bytes out of the read path. + const float * lat_spk_emb = (params->abi_version >= 2) ? params->ref_spk_emb : NULL; + const int lat_spk_dim = (params->abi_version >= 2) ? params->ref_spk_dim : 0; + const int32_t * lat_codes = (params->abi_version >= 2) ? params->ref_codes : NULL; + const int lat_T = (params->abi_version >= 2) ? params->ref_T : 0; + + const bool has_ref_audio = (params->ref_audio_24k != NULL) && (params->ref_n_samples > 0); + const bool has_lat_spk = (lat_spk_emb != NULL) && (lat_spk_dim > 0); + const bool has_lat_codes = (lat_codes != NULL) && (lat_T > 0); + + // Raw waveform and pre-encoded latents are mutually exclusive: the + // caller is told immediately rather than picking a winner silently. + if (has_ref_audio && (has_lat_spk || has_lat_codes)) { + qt_set_error("pipeline_tts_synthesize: ref_audio_24k and ref_spk_emb / ref_codes are mutually exclusive"); + qt_log(QT_LOG_ERROR, "[Pipeline] ref_audio_24k and ref_spk_emb / ref_codes are mutually exclusive"); + return QT_STATUS_INVALID_PARAMS; + } + // Latent ICL codes ride on top of the speaker embedding and need the + // transcript, mirroring the raw path where mode B implies mode A. + if (has_lat_codes && (!has_lat_spk || ref_text.empty())) { + qt_set_error("pipeline_tts_synthesize: ref_codes requires ref_spk_emb and ref_text"); + qt_log(QT_LOG_ERROR, "[Pipeline] ref_codes requires ref_spk_emb and ref_text"); + return QT_STATUS_INVALID_PARAMS; + } + + // Voice clone mode A: a pre-extracted latent embedding feeds the + // prompt builder directly; otherwise, if ref_audio_24k is given, run + // the speaker encoder on the pre-decoded mono buffer. Mutually + // exclusive with --speaker. std::vector ref_spk_emb; const float * ref_spk_emb_ptr = NULL; - if (has_ref_audio) { + if (has_lat_spk) { + if (lat_spk_dim != pt->talker.hidden_size) { + qt_set_error("pipeline_tts_synthesize: ref_spk_dim %d mismatches talker hidden %d", lat_spk_dim, + pt->talker.hidden_size); + qt_log(QT_LOG_ERROR, "[Pipeline] ref_spk_dim %d mismatches talker hidden %d", lat_spk_dim, + pt->talker.hidden_size); + return QT_STATUS_INVALID_PARAMS; + } + ref_spk_emb_ptr = lat_spk_emb; + qt_log(QT_LOG_INFO, "[Pipeline] Latent speaker embedding: %d values", lat_spk_dim); + } else if (has_ref_audio) { if (!pt->has_speaker_encoder) { qt_set_error( "pipeline_tts_synthesize: --ref-wav requires a model with a loaded speaker encoder (Base only)"); @@ -404,17 +440,22 @@ qt_status pipeline_tts_synthesize(PipelineTTS * pt, ref_spk_emb_ptr = ref_spk_emb.data(); } - // Voice clone mode B: if ref_text is also given, encode the - // reference audio into 16 codebook indices via the codec encoder. - // Layout returned by pipeline_codec_encode is [num_codebooks, T_codec] - // row major, matching what the prompt builder expects for the ICL - // sum loop. + // Voice clone mode B: pre-encoded latent codes feed the ICL prompt + // directly; otherwise, if ref_text is given, encode the reference + // audio into 16 codebook indices via the codec encoder. Layout is + // [num_codebooks, T_codec] row major in both cases, matching what + // the prompt builder expects for the ICL sum loop. std::vector ref_codes; - int ref_codes_T = 0; - if (!ref_text.empty()) { + const int32_t * ref_codes_ptr = NULL; + int ref_codes_T = 0; + if (has_lat_codes) { + ref_codes_ptr = lat_codes; + ref_codes_T = lat_T; + qt_log(QT_LOG_INFO, "[Pipeline] Latent ICL ref_codes: %d frames at 12.5 Hz", ref_codes_T); + } else if (!ref_text.empty()) { if (!has_ref_audio) { - qt_set_error("pipeline_tts_synthesize: --ref-text requires --ref-wav"); - qt_log(QT_LOG_ERROR, "[Pipeline] --ref-text requires --ref-wav"); + qt_set_error("pipeline_tts_synthesize: ref_text requires ref_audio_24k or latent ref_codes"); + qt_log(QT_LOG_ERROR, "[Pipeline] ref_text requires ref_audio_24k or latent ref_codes"); return QT_STATUS_INVALID_PARAMS; } // The codec hop is 1920 samples at 24 kHz so n_samples must be @@ -431,7 +472,8 @@ qt_status pipeline_tts_synthesize(PipelineTTS * pt, qt_log(QT_LOG_ERROR, "[Pipeline] pipeline_codec_encode returned empty codes"); return QT_STATUS_GENERATE_FAILED; } - ref_codes_T = (int) ref_codes.size() / pt->num_code_groups; + ref_codes_ptr = ref_codes.data(); + ref_codes_T = (int) ref_codes.size() / pt->num_code_groups; qt_log(QT_LOG_INFO, "[Pipeline] ICL ref_codes: %d frames at 12.5 Hz (%d audio samples)", ref_codes_T, aligned_T); } @@ -444,8 +486,8 @@ qt_status pipeline_tts_synthesize(PipelineTTS * pt, const char * lang = params->lang ? params->lang : "auto"; Timer t_build; - if (!prompt_builder_build(pt, tok, params->text, lang, instruct, speaker, ref_spk_emb_ptr, ref_text, - ref_codes_T > 0 ? ref_codes.data() : NULL, ref_codes_T, &prompt)) { + if (!prompt_builder_build(pt, tok, params->text, lang, instruct, speaker, ref_spk_emb_ptr, ref_text, ref_codes_ptr, + ref_codes_T, &prompt)) { return QT_STATUS_GENERATE_FAILED; } perf.build_ms = t_build.ms(); @@ -469,7 +511,7 @@ qt_status pipeline_tts_synthesize(PipelineTTS * pt, } if (ref_codes_T > 0) { const int shape[2] = { pt->num_code_groups, ref_codes_T }; - debug_dump_i32_as_f32(&d, "ref-codes", ref_codes.data(), shape, 2); + debug_dump_i32_as_f32(&d, "ref-codes", ref_codes_ptr, shape, 2); } } diff --git a/src/qwen.cpp b/src/qwen.cpp index 038622a..0bc3615 100644 --- a/src/qwen.cpp +++ b/src/qwen.cpp @@ -218,6 +218,18 @@ void qt_tts_default_params(struct qt_tts_params * p) { p->on_chunk_user_data = nullptr; p->codec_chunk_sec = 24.0f; p->codec_left_context_sec = 2.0f; + p->ref_spk_emb = nullptr; + p->ref_spk_dim = 0; + p->ref_codes = nullptr; + p->ref_T = 0; +} + +int qt_num_codebooks(const struct qt_context * q) { + if (!q) { + qt_set_error("qt_num_codebooks: q is NULL"); + return 0; + } + return q->pt.num_code_groups; } struct qt_context * qt_init(const struct qt_init_params * params) { @@ -356,22 +368,26 @@ enum qt_status qt_synthesize(struct qt_context * q, const struct qt_tts_params * } return QT_STATUS_MODE_INVALID; } - if (params->ref_audio_24k && mt != "base") { - qt_set_error("--ref-wav is only valid for base models (loaded: %s)", mt.c_str()); + // ABI v2 latent reference fields, same gate as the pipeline. + const bool has_lat_spk = params->abi_version >= 2 && params->ref_spk_emb && params->ref_spk_dim > 0; + const bool has_lat_codes = params->abi_version >= 2 && params->ref_codes && params->ref_T > 0; + + if ((params->ref_audio_24k || has_lat_spk) && mt != "base") { + qt_set_error("--ref-wav / --ref-spk is only valid for base models (loaded: %s)", mt.c_str()); if (out) { qt_audio_free(out); } return QT_STATUS_MODE_INVALID; } - if (params->speaker && params->ref_audio_24k) { - qt_set_error("--speaker and --ref-wav are mutually exclusive"); + if (params->speaker && (params->ref_audio_24k || has_lat_spk)) { + qt_set_error("--speaker and --ref-wav / --ref-spk are mutually exclusive"); if (out) { qt_audio_free(out); } return QT_STATUS_INVALID_PARAMS; } - if (params->ref_text && !params->ref_audio_24k) { - qt_set_error("--ref-text requires --ref-wav"); + if (params->ref_text && !params->ref_audio_24k && !has_lat_codes) { + qt_set_error("--ref-text requires --ref-wav or --ref-rvq"); if (out) { qt_audio_free(out); } diff --git a/src/qwen.h b/src/qwen.h index 88d78ec..80d55d9 100644 --- a/src/qwen.h +++ b/src/qwen.h @@ -57,7 +57,7 @@ extern "C" { // git short hash + commit date string returned by qt_version(); for // binding compat checks, QT_ABI_VERSION is the only number that // matters. -#define QT_ABI_VERSION 1 +#define QT_ABI_VERSION 2 // Returns a static string of the form " ()" identifying // the exact commit this binary was built from. Safe to call from any @@ -269,6 +269,19 @@ struct qt_tts_params { // clamps to >= 0 frames. float codec_chunk_sec; float codec_left_context_sec; + + // ABI v2. Pre-encoded voice reference, the latent counterpart of + // ref_audio_24k. ref_spk_emb is the speaker embedding produced by + // the speaker encoder (ref_spk_dim f32 values, must equal the + // talker hidden size). ref_codes is the ICL code matrix produced + // by the codec encoder, [num_codebooks, ref_T] row-major. + // ref_spk_emb alone selects clone mode A; ref_spk_emb + ref_codes + // + ref_text selects mode B, mirroring the raw constraints. + // Mutually exclusive with ref_audio_24k and speaker. + const float * ref_spk_emb; + int ref_spk_dim; + const int32_t * ref_codes; + int ref_T; }; // Initialise to the standard defaults. Strings NULL, seed -1, @@ -278,6 +291,12 @@ struct qt_tts_params { // codec_left_context_sec 2.0. QT_API void qt_tts_default_params(struct qt_tts_params * p); +// Number of RVQ codebooks (K) of the loaded codec. Pre-encoded ICL +// reference codes passed via ref_codes are laid out [K, ref_T] +// row-major; callers reading a packed .rvq stream need K to derive +// ref_T from the code count. Returns 0 on a NULL handle. +QT_API int qt_num_codebooks(const struct qt_context * q); + // Run the full TTS synthesis. Validates the params against the loaded // model_type (the seven base / custom_voice / voice_design rules), // resolves the seed, hands off to pipeline_tts_synthesize and fills diff --git a/src/rvq-file.h b/src/rvq-file.h new file mode 100644 index 0000000..b152db7 --- /dev/null +++ b/src/rvq-file.h @@ -0,0 +1,111 @@ +#pragma once +// rvq-file.h: packed RVQ code stream file IO (.rvq). +// +// Flat code stream packed at code_bits per code, LSB-first, no header. +// Layout is [K, T] row-major. K and code_bits are fixed by the codec +// config in the GGUF; T is derived from the file size: +// T = (filesize * 8) / (K * code_bits). + +#include "utf8.h" + +#include +#include +#include +#include + +// Pack a flat code stream into code_bits-per-code, LSB-first. Output size +// is ceil(N * code_bits / 8) bytes. +static std::vector rvq_pack_codes(const std::vector & codes, int code_bits) { + const uint32_t mask = (1u << code_bits) - 1u; + const size_t total_bits = codes.size() * (size_t) code_bits; + std::vector out((total_bits + 7) / 8, 0); + + uint64_t acc = 0; + int bits_in_acc = 0; + size_t out_pos = 0; + for (size_t i = 0; i < codes.size(); i++) { + acc |= ((uint64_t) ((uint32_t) codes[i] & mask)) << bits_in_acc; + bits_in_acc += code_bits; + while (bits_in_acc >= 8) { + out[out_pos++] = (uint8_t) (acc & 0xFF); + acc >>= 8; + bits_in_acc -= 8; + } + } + if (bits_in_acc > 0) { + out[out_pos++] = (uint8_t) (acc & 0xFF); + } + return out; +} + +// Symmetric unpack: reads N codes from packed bytes. +static std::vector rvq_unpack_codes(const std::vector & in, size_t n_codes, int code_bits) { + const uint32_t mask = (1u << code_bits) - 1u; + std::vector out(n_codes); + + uint64_t acc = 0; + int bits_in_acc = 0; + size_t in_pos = 0; + for (size_t i = 0; i < n_codes; i++) { + while (bits_in_acc < code_bits && in_pos < in.size()) { + acc |= ((uint64_t) in[in_pos++]) << bits_in_acc; + bits_in_acc += 8; + } + out[i] = (int32_t) (acc & mask); + acc >>= code_bits; + bits_in_acc -= code_bits; + } + return out; +} + +// Read a .rvq file and unpack it into K*T codes. T is inferred from the +// file size. +static bool rvq_read_file(const char * path, int K, int code_bits, std::vector & codes, int * n_frames) { + FILE * f = utf8_fopen(path, "rb"); + if (!f) { + fprintf(stderr, "[RVQ] FATAL: cannot open %s\n", path); + return false; + } + fseek(f, 0, SEEK_END); + long sz = ftell(f); + fseek(f, 0, SEEK_SET); + if (sz <= 0) { + fprintf(stderr, "[RVQ] FATAL: %s is empty\n", path); + fclose(f); + return false; + } + std::vector buf((size_t) sz); + if (fread(buf.data(), 1, buf.size(), f) != buf.size()) { + fprintf(stderr, "[RVQ] FATAL: short read on %s\n", path); + fclose(f); + return false; + } + fclose(f); + + const size_t total_bits = (size_t) sz * 8; + const size_t n_codes = total_bits / (size_t) code_bits; + if (n_codes == 0 || (n_codes % (size_t) K) != 0) { + fprintf(stderr, "[RVQ] FATAL: %s yields %zu codes, not a multiple of K=%d\n", path, n_codes, K); + return false; + } + codes = rvq_unpack_codes(buf, n_codes, code_bits); + *n_frames = (int) (n_codes / (size_t) K); + return true; +} + +// Pack and write a .rvq file. +static bool rvq_write_file(const char * path, const std::vector & codes, int code_bits) { + std::vector packed = rvq_pack_codes(codes, code_bits); + FILE * f = utf8_fopen(path, "wb"); + if (!f) { + fprintf(stderr, "[RVQ] FATAL: cannot open %s for write\n", path); + return false; + } + if (fwrite(packed.data(), 1, packed.size(), f) != packed.size()) { + fprintf(stderr, "[RVQ] FATAL: short write on %s\n", path); + fclose(f); + return false; + } + fclose(f); + return true; +} diff --git a/tools/qwen-codec.cpp b/tools/qwen-codec.cpp index 3aeb15f..1599da7 100644 --- a/tools/qwen-codec.cpp +++ b/tools/qwen-codec.cpp @@ -5,6 +5,13 @@ // file extension: .wav in -> encode, .rvq in -> decode. Output is // auto-named next to the input file by swapping the extension. // +// Encode truncates the input to the hop boundary, strictly conforming +// to the qwen-tts --ref-wav ICL path, so a .rvq produced here feeds +// qwen-tts --ref-rvq directly. Passing --talker additionally runs the +// speaker encoder from the talker GGUF on the full input and writes +// the x-vector embedding next to the .rvq as a .spk file (raw f32, +// enc_dim values), feeding qwen-tts --ref-spk. +// // File format (.rvq): flat code stream packed at 11 bits per code, // LSB-first, no header. Layout is [K, T] row-major. K is fixed by the // codec config in the GGUF (16 codebooks for the 12Hz tokenizer, @@ -12,7 +19,11 @@ #include "audio-io.h" #include "backend.h" +#include "gguf-weights.h" #include "pipeline-codec.h" +#include "rvq-file.h" +#include "speaker-encoder-extract.h" +#include "speaker-encoder-weights.h" #include "utf8.h" #include "version.h" @@ -23,115 +34,24 @@ #include #include -static const uint32_t RVQ_CODE_MASK = (1u << TOKENIZER_CODE_BITS) - 1u; - static void print_usage(const char * prog) { fprintf(stderr, "qwentts.cpp %s\n\n", QWEN_VERSION); fprintf(stderr, - "Usage: %s --model [-i ] [--format ]\n\n" + "Usage: %s --model [-i ] [--talker ] [--format ]\n\n" "Required:\n" " --model Codec GGUF (qwen-tokenizer-12hz-*.gguf)\n\n" "Optional:\n" " -i Input. WAV -> encode, .rvq -> decode\n" + " --talker Talker GGUF (Base only). Encode also extracts the speaker\n" + " embedding and writes it next to the .rvq as a .spk file\n" " --format WAV output format: wav16, wav24, wav32 (default: wav16)\n\n" "Output is auto-named next to input : clip.wav -> clip.rvq, clip.rvq -> clip.wav.\n" + "Encode truncates to the hop boundary, conforming to the qwen-tts --ref-wav path:\n" + "the .rvq feeds qwen-tts --ref-rvq, the .spk feeds qwen-tts --ref-spk.\n" "When -i is omitted, runs a load self-test of the codec GGUF.\n", prog); } -// Symmetric unpack: reads N codes from packed bytes (11 bits LSB-first). -static std::vector unpack_codes(const std::vector & in, size_t n_codes) { - std::vector out(n_codes); - uint64_t acc = 0; - int bits_in_acc = 0; - size_t in_pos = 0; - for (size_t i = 0; i < n_codes; i++) { - while (bits_in_acc < TOKENIZER_CODE_BITS && in_pos < in.size()) { - acc |= ((uint64_t) in[in_pos++]) << bits_in_acc; - bits_in_acc += 8; - } - out[i] = (int32_t) (acc & RVQ_CODE_MASK); - acc >>= TOKENIZER_CODE_BITS; - bits_in_acc -= TOKENIZER_CODE_BITS; - } - return out; -} - -// Pack flat int32 codes into 11-bit LSB-first packed bytes. Output size is -// ceil(N * 11 / 8) bytes. -static std::vector pack_codes(const std::vector & codes) { - const size_t total_bits = codes.size() * (size_t) TOKENIZER_CODE_BITS; - std::vector out((total_bits + 7) / 8, 0); - uint64_t acc = 0; - int bits_in_acc = 0; - size_t out_pos = 0; - for (size_t i = 0; i < codes.size(); i++) { - acc |= ((uint64_t) ((uint32_t) codes[i] & RVQ_CODE_MASK)) << bits_in_acc; - bits_in_acc += TOKENIZER_CODE_BITS; - while (bits_in_acc >= 8) { - out[out_pos++] = (uint8_t) (acc & 0xFF); - acc >>= 8; - bits_in_acc -= 8; - } - } - if (bits_in_acc > 0) { - out[out_pos++] = (uint8_t) (acc & 0xFF); - } - return out; -} - -// Read a .rvq file and unpack it into K*T codes. T is inferred from the -// file size: T = (filesize * 8) / (K * TOKENIZER_CODE_BITS). -static bool read_rvq(const char * path, int K, std::vector & codes, int * n_frames) { - FILE * f = utf8_fopen(path, "rb"); - if (!f) { - fprintf(stderr, "[Codec] FATAL: cannot open %s\n", path); - return false; - } - fseek(f, 0, SEEK_END); - long sz = ftell(f); - fseek(f, 0, SEEK_SET); - if (sz <= 0) { - fprintf(stderr, "[Codec] FATAL: %s is empty\n", path); - fclose(f); - return false; - } - std::vector buf((size_t) sz); - if (fread(buf.data(), 1, buf.size(), f) != buf.size()) { - fprintf(stderr, "[Codec] FATAL: short read on %s\n", path); - fclose(f); - return false; - } - fclose(f); - - const size_t total_bits = (size_t) sz * 8; - const size_t n_codes = total_bits / (size_t) TOKENIZER_CODE_BITS; - if (n_codes == 0 || (n_codes % (size_t) K) != 0) { - fprintf(stderr, "[Codec] FATAL: %s yields %zu codes, not a multiple of K=%d\n", path, n_codes, K); - return false; - } - codes = unpack_codes(buf, n_codes); - *n_frames = (int) (n_codes / (size_t) K); - return true; -} - -// Pack and write a .rvq file. -static bool write_rvq(const char * path, const std::vector & codes) { - std::vector packed = pack_codes(codes); - FILE * f = utf8_fopen(path, "wb"); - if (!f) { - fprintf(stderr, "[Codec] FATAL: cannot open %s for write\n", path); - return false; - } - if (fwrite(packed.data(), 1, packed.size(), f) != packed.size()) { - fprintf(stderr, "[Codec] FATAL: short write on %s\n", path); - fclose(f); - return false; - } - fclose(f); - return true; -} - // Replace or append extension on a path string. static std::string swap_ext(const std::string & path, const char * ext) { size_t dot = path.find_last_of('.'); @@ -154,6 +74,60 @@ static int infer_mode(const char * path) { return 0; } +// Load the speaker encoder from the talker GGUF, run it on the full input +// buffer and write the embedding as a raw f32 .spk file (enc_dim values, +// validated by filesize on the qwen-tts side). Returns 0 on success. +static int extract_spk(const char * talker_path, + BackendPair bp, + const float * audio, + int n_samples, + const char * out_path) { + GGUFModel gf = {}; + if (!gf_load(&gf, talker_path)) { + fprintf(stderr, "[Codec] FATAL: cannot open talker GGUF %s\n", talker_path); + return 1; + } + + SpeakerEncoderWeights sw = {}; + if (!speaker_encoder_weights_load(&sw, gf, bp.backend)) { + fprintf(stderr, "[Codec] FATAL: speaker encoder load failed from %s\n", talker_path); + gf_close(&gf); + return 1; + } + gf_close(&gf); + if (sw.weight_buf == NULL) { + fprintf(stderr, "[Codec] FATAL: %s has no speaker encoder (Base only)\n", talker_path); + return 1; + } + + ggml_backend_sched_t sched = backend_sched_new(bp, 4096); + const int enc_dim = sw.enc_dim; + std::vector emb; + bool ok = speaker_encoder_extract(&sw, sched, audio, n_samples, emb); + ggml_backend_sched_free(sched); + speaker_encoder_weights_free(&sw); + if (!ok || (int) emb.size() != enc_dim) { + fprintf(stderr, "[Codec] FATAL: speaker embedding extraction failed (%zu values, enc_dim %d)\n", emb.size(), + enc_dim); + return 1; + } + + FILE * f = utf8_fopen(out_path, "wb"); + if (!f) { + fprintf(stderr, "[Codec] FATAL: cannot open %s for write\n", out_path); + return 1; + } + if (fwrite(emb.data(), sizeof(float), emb.size(), f) != emb.size()) { + fprintf(stderr, "[Codec] FATAL: short write on %s\n", out_path); + fclose(f); + return 1; + } + fclose(f); + + fprintf(stderr, "[Codec] Wrote %s: %zu f32 values (%zu bytes)\n", out_path, emb.size(), emb.size() * sizeof(float)); + return 0; +} + int main(int argc, char ** argv) { utf8_init(&argc, &argv); if (argc <= 1) { @@ -161,13 +135,16 @@ int main(int argc, char ** argv) { return 0; } - const char * model_path = NULL; - const char * input_path = NULL; - WavFormat wav_fmt = WAV_S16; + const char * model_path = NULL; + const char * input_path = NULL; + const char * talker_path = NULL; + WavFormat wav_fmt = WAV_S16; for (int i = 1; i < argc; i++) { if (strcmp(argv[i], "--model") == 0 && i + 1 < argc) { model_path = argv[++i]; + } else if (strcmp(argv[i], "--talker") == 0 && i + 1 < argc) { + talker_path = argv[++i]; } else if (strcmp(argv[i], "-i") == 0 && i + 1 < argc) { input_path = argv[++i]; } else if (strcmp(argv[i], "--format") == 0 && i + 1 < argc) { @@ -217,40 +194,46 @@ int main(int argc, char ** argv) { if (!input_path) { fprintf(stderr, "[Codec] Load self-test passed\n"); } else if (mode == 1) { - // Encode .wav -> .rvq + // Encode .wav -> .rvq (+ .spk with --talker) const std::string out_str = swap_ext(input_path, ".rvq"); int T_in = 0; float * audio_in = audio_read_mono(input_path, TOKENIZER_SAMPLE_RATE, &T_in); - if (!audio_in || T_in <= 0) { - fprintf(stderr, "[Codec] FATAL: cannot read %s\n", input_path); + if (!audio_in || T_in < TOKENIZER_HOP_LENGTH) { + fprintf(stderr, "[Codec] FATAL: cannot read %s or input shorter than one hop (%d samples)\n", input_path, + TOKENIZER_HOP_LENGTH); free(audio_in); rc = 1; } else { - // Pad to a multiple of HOP_LENGTH so the RVQ frame count is integral. - int hop = TOKENIZER_HOP_LENGTH; - int T_padded = ((T_in + hop - 1) / hop) * hop; - int T_frames = T_padded / hop; + // Truncate to a multiple of HOP_LENGTH, strictly conforming to + // the qwen-tts --ref-wav ICL path. + int hop = TOKENIZER_HOP_LENGTH; + int T_aligned = (T_in / hop) * hop; + int T_frames = T_aligned / hop; - std::vector audio_buf((size_t) T_padded, 0.0f); - memcpy(audio_buf.data(), audio_in, (size_t) T_in * sizeof(float)); - free(audio_in); + fprintf(stderr, "[Codec] Encode: %s, %d samples @ %d Hz, truncated to %d (%d frames @ 12.5 Hz, %.2f s)\n", + input_path, T_in, TOKENIZER_SAMPLE_RATE, T_aligned, T_frames, + (double) T_aligned / (double) TOKENIZER_SAMPLE_RATE); - fprintf(stderr, "[Codec] Encode: %s, %d samples @ %d Hz, padded to %d (%d frames @ 12.5 Hz, %.2f s)\n", - input_path, T_in, TOKENIZER_SAMPLE_RATE, T_padded, T_frames, - (double) T_padded / (double) TOKENIZER_SAMPLE_RATE); - - std::vector codes = pipeline_codec_encode(&pc, audio_buf.data(), T_padded); + std::vector codes = pipeline_codec_encode(&pc, audio_in, T_aligned); if (codes.empty()) { fprintf(stderr, "[Codec] FATAL: encode failed\n"); rc = 1; - } else if (!write_rvq(out_str.c_str(), codes)) { + } else if (!rvq_write_file(out_str.c_str(), codes, TOKENIZER_CODE_BITS)) { rc = 1; } else { fprintf(stderr, "[Codec] Wrote %s: K=%d T=%d, %zu codes -> %zu packed bytes\n", out_str.c_str(), TOKENIZER_NUM_CODEBOOKS, T_frames, codes.size(), (codes.size() * (size_t) TOKENIZER_CODE_BITS + 7) / 8); } + + // Speaker embedding extraction, conforming to the qwen-tts + // --ref-wav mode A path: the encoder consumes the FULL input + // buffer, never the hop-truncated one. + if (rc == 0 && talker_path) { + rc = extract_spk(talker_path, bp, audio_in, T_in, swap_ext(input_path, ".spk").c_str()); + } + free(audio_in); } } else { // Decode .rvq -> .wav @@ -258,7 +241,7 @@ int main(int argc, char ** argv) { std::vector codes; int T = 0; - if (!read_rvq(input_path, TOKENIZER_NUM_CODEBOOKS, codes, &T)) { + if (!rvq_read_file(input_path, TOKENIZER_NUM_CODEBOOKS, TOKENIZER_CODE_BITS, codes, &T)) { rc = 1; } else { fprintf(stderr, "[Codec] Decode: %s, K=%d T=%d (%.2f s)\n", input_path, TOKENIZER_NUM_CODEBOOKS, T, diff --git a/tools/qwen-tts.cpp b/tools/qwen-tts.cpp index 95ec2c2..f07f653 100644 --- a/tools/qwen-tts.cpp +++ b/tools/qwen-tts.cpp @@ -13,6 +13,7 @@ #include "audio-io.h" #include "qwen.h" +#include "rvq-file.h" #include #include @@ -40,6 +41,10 @@ static void print_usage(const char * prog) { " CustomVoice, rejected for Base\n" " --speaker Speaker name (CustomVoice only)\n" " --ref-wav Reference WAV for voice cloning (Base only)\n" + " --ref-spk Pre-extracted speaker embedding from qwen-codec --talker\n" + " (replaces --ref-wav, Base only)\n" + " --ref-rvq Pre-encoded reference codes from qwen-codec (requires\n" + " --ref-spk and --ref-text, enables ICL clone mode)\n" " --ref-text Transcript file for the reference (enables ICL clone mode)\n" " --max-new Max new audio frames (default: 2048)\n" " --codec-chunk-dur Codec decode chunk duration in seconds (default: 24.0)\n" @@ -69,6 +74,8 @@ struct Args { const char * instruct; const char * speaker; const char * ref_wav; + const char * ref_spk; + const char * ref_rvq; const char * ref_text_path; const char * dump_dir; const char * out_wav; @@ -104,6 +111,34 @@ static std::string read_stdin_text() { } // Read a small text file into a string. Trims trailing newlines. +// 11 bits per code (V <= 2048), matching qwen-codec. +static const int RVQ_CODE_BITS = 11; + +// Read a .spk file: raw f32 values, the count IS the embedding dimension. +static bool read_spk_file(const char * path, std::vector & emb) { + FILE * f = utf8_fopen(path, "rb"); + if (!f) { + fprintf(stderr, "[CLI] ERROR: cannot open --ref-spk '%s'\n", path); + return false; + } + fseek(f, 0, SEEK_END); + long sz = ftell(f); + fseek(f, 0, SEEK_SET); + if (sz <= 0 || (sz % (long) sizeof(float)) != 0) { + fprintf(stderr, "[CLI] ERROR: --ref-spk '%s' size %ld is not a positive multiple of 4\n", path, sz); + fclose(f); + return false; + } + emb.resize((size_t) sz / sizeof(float)); + if (fread(emb.data(), sizeof(float), emb.size(), f) != emb.size()) { + fprintf(stderr, "[CLI] ERROR: short read on --ref-spk '%s'\n", path); + fclose(f); + return false; + } + fclose(f); + return true; +} + static bool read_text_file(const char * path, std::string & out) { FILE * f = fopen(path, "rb"); if (!f) { @@ -168,6 +203,10 @@ static bool parse_args(int argc, char ** argv, Args & a) { a.speaker = argv[++i]; } else if (std::strcmp(arg, "--ref-wav") == 0 && i + 1 < argc) { a.ref_wav = argv[++i]; + } else if (std::strcmp(arg, "--ref-spk") == 0 && i + 1 < argc) { + a.ref_spk = argv[++i]; + } else if (std::strcmp(arg, "--ref-rvq") == 0 && i + 1 < argc) { + a.ref_rvq = argv[++i]; } else if (std::strcmp(arg, "--ref-text") == 0 && i + 1 < argc) { a.ref_text_path = argv[++i]; } else if (std::strcmp(arg, "--format") == 0 && i + 1 < argc) { @@ -278,6 +317,29 @@ static int run(const Args & a) { ref_n_samples = T_in; } + // Latent reference files. The .spk holds raw f32 values whose count + // IS the embedding dimension; the .rvq holds the packed ICL code + // matrix. The facade validates the structural constraints (mutual + // exclusions, dim match against the talker hidden size). + std::vector ref_spk_emb; + std::vector ref_codes; + int ref_T = 0; + if (a.ref_spk) { + if (!read_spk_file(a.ref_spk, ref_spk_emb)) { + qt_free(q); + return 1; + } + fprintf(stderr, "[CLI] Reference SPK: %s, %zu f32 values\n", a.ref_spk, ref_spk_emb.size()); + } + if (a.ref_rvq) { + const int K = qt_num_codebooks(q); + if (!rvq_read_file(a.ref_rvq, K, RVQ_CODE_BITS, ref_codes, &ref_T)) { + qt_free(q); + return 1; + } + fprintf(stderr, "[CLI] Reference RVQ: %s, K=%d T=%d\n", a.ref_rvq, K, ref_T); + } + // Resolve output WAV format string: wav16 / wav24 / wav32. Default // wav16 mirrors the omnivoice.cpp default. WavFormat wav_fmt; @@ -323,6 +385,10 @@ static int run(const Args & a) { params.ref_audio_24k = ref_audio_24k; params.ref_n_samples = ref_n_samples; params.ref_text = ref_text; + params.ref_spk_emb = ref_spk_emb.empty() ? NULL : ref_spk_emb.data(); + params.ref_spk_dim = (int) ref_spk_emb.size(); + params.ref_codes = ref_codes.empty() ? NULL : ref_codes.data(); + params.ref_T = ref_T; params.seed = a.seed; params.max_new_tokens = a.max_new_tokens; params.do_sample = a.do_sample;