HelloWorld2307 commited on 24 days ago

Commit

995e681

verified ·

1 Parent(s): 1b0cd2e

Upload 2flow folder

Browse files

Files changed (32) hide show

2flow/Dockerfile +3 -0
2flow/models/.gitattributes +35 -0
2flow/models/README.md +20 -0
2flow/models/downloads/F5TTS_Base/model_1200000.pt +3 -0
2flow/models/downloads/F5TTS_Base/model_1200000.safetensors +3 -0
2flow/models/downloads/F5TTS_Base/vocab.txt +2545 -0
2flow/models/downloads/F5TTS_Base_bigvgan/model_1250000.pt +3 -0
2flow/models/downloads/F5TTS_v1_Base/model_1250000.safetensors +3 -0
2flow/models/downloads/F5TTS_v1_Base/vocab.txt +2545 -0
2flow/models/downloads/F5TTS_v1_Base_no_zero_init/model_1250000.safetensors +3 -0
2flow/patch/__init__.py +196 -0
2flow/patch/f5tts/model.py +222 -0
2flow/patch/f5tts/modules.py +447 -0
2flow/requirements.txt +5 -0
2flow/scripts/build.sh +2 -0
2flow/scripts/f5/build_engine.sh +5 -0
2flow/scripts/f5/fix_lib.py +32 -0
2flow/scripts/f5/pre_build_engine.sh +4 -0
2flow/scripts/init.sh +6 -0
2flow/scripts/vocoder/build_engine.sh +3 -0
2flow/scripts/vocoder/export_vocos_trt.sh +43 -0
2flow/scripts/vocoder/pre_build_engine.sh +3 -0
2flow/services/triton/f5_tts_triton_server/f5_tts/1/f5_tts_trtllm.py +486 -0
2flow/services/triton/f5_tts_triton_server/f5_tts/1/model.py +278 -0
2flow/services/triton/f5_tts_triton_server/f5_tts/config.pbtxt +81 -0
2flow/services/triton/f5_tts_triton_server/vocoder/1/.gitkeep +0 -0
2flow/services/triton/f5_tts_triton_server/vocoder/config.pbtxt +32 -0
2flow/utils/tts/__pycache__/convert_checkpoint.cpython-310.pyc +0 -0
2flow/utils/tts/__pycache__/convert_checkpoint.cpython-312.pyc +0 -0
2flow/utils/tts/__pycache__/export_vocoder_to_onnx.cpython-312.pyc +0 -0
2flow/utils/tts/convert_checkpoint.py +378 -0
2flow/utils/tts/export_vocoder_to_onnx.py +138 -0

2flow/Dockerfile ADDED Viewed

	@@ -0,0 +1,3 @@

+FROM nvcr.io/nvidia/tritonserver:25.04-py3
+WORKDIR /workspace/2flow
+COPY . .

2flow/models/.gitattributes ADDED Viewed

	@@ -0,0 +1,35 @@

+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text

2flow/models/README.md ADDED Viewed

	@@ -0,0 +1,20 @@

+---
+license: cc-by-nc-4.0
+pipeline_tag: text-to-speech
+library_name: f5-tts
+datasets:
+- amphion/Emilia-Dataset
+---
+Download [F5-TTS](https://huggingface.co/SWivid/F5-TTS/tree/main/F5TTS_Base) or [E2 TTS](https://huggingface.co/SWivid/E2-TTS/tree/main/E2TTS_Base) and place under ckpts/
+```
+ckpts/
+    F5TTS_v1_Base/
+        model_1250000.safetensors
+    F5TTS_Base/
+        model_1200000.safetensors
+    E2TTS_Base/
+        model_1200000.safetensors
+```
+Github: https://github.com/SWivid/F5-TTS
+Paper: [F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching](https://huggingface.co/papers/2410.06885)

2flow/models/downloads/F5TTS_Base/model_1200000.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c2f1bcbe1582a04468920abf227aa75f18faf57d24d5b141195eb4e55f39bc03
+size 1348767810

2flow/models/downloads/F5TTS_Base/model_1200000.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4180310f91d592cee4bc14998cd37c781f779cf105e8ca8744d9bd48ca7046ae
+size 1348645281

2flow/models/downloads/F5TTS_Base/vocab.txt ADDED Viewed

	@@ -0,0 +1,2545 @@

+!
+"
+#
+$
+%
+&
+'
+(
+)
+*
++
+,
+-
+.
+/
+0
+1
+2
+3
+4
+5
+6
+7
+8
+9
+:
+;
+=
+>
+?
+@
+A
+B
+C
+D
+E
+F
+G
+H
+I
+J
+K
+L
+M
+N
+O
+P
+Q
+R
+S
+T
+U
+V
+W
+X
+Y
+Z
+[
+\
+]
+_
+a
+a1
+ai1
+ai2
+ai3
+ai4
+an1
+an3
+an4
+ang1
+ang2
+ang4
+ao1
+ao2
+ao3
+ao4
+b
+ba
+ba1
+ba2
+ba3
+ba4
+bai1
+bai2
+bai3
+bai4
+ban1
+ban2
+ban3
+ban4
+bang1
+bang2
+bang3
+bang4
+bao1
+bao2
+bao3
+bao4
+bei
+bei1
+bei2
+bei3
+bei4
+ben1
+ben2
+ben3
+ben4
+beng
+beng1
+beng2
+beng3
+beng4
+bi1
+bi2
+bi3
+bi4
+bian1
+bian2
+bian3
+bian4
+biao1
+biao2
+biao3
+bie1
+bie2
+bie3
+bie4
+bin1
+bin4
+bing1
+bing2
+bing3
+bing4
+bo
+bo1
+bo2
+bo3
+bo4
+bu2
+bu3
+bu4
+c
+ca1
+cai1
+cai2
+cai3
+cai4
+can1
+can2
+can3
+can4
+cang1
+cang2
+cao1
+cao2
+cao3
+ce4
+cen1
+cen2
+ceng1
+ceng2
+ceng4
+cha1
+cha2
+cha3
+cha4
+chai1
+chai2
+chan1
+chan2
+chan3
+chan4
+chang1
+chang2
+chang3
+chang4
+chao1
+chao2
+chao3
+che1
+che2
+che3
+che4
+chen1
+chen2
+chen3
+chen4
+cheng1
+cheng2
+cheng3
+cheng4
+chi1
+chi2
+chi3
+chi4
+chong1
+chong2
+chong3
+chong4
+chou1
+chou2
+chou3
+chou4
+chu1
+chu2
+chu3
+chu4
+chua1
+chuai1
+chuai2
+chuai3
+chuai4
+chuan1
+chuan2
+chuan3
+chuan4
+chuang1
+chuang2
+chuang3
+chuang4
+chui1
+chui2
+chun1
+chun2
+chun3
+chuo1
+chuo4
+ci1
+ci2
+ci3
+ci4
+cong1
+cong2
+cou4
+cu1
+cu4
+cuan1
+cuan2
+cuan4
+cui1
+cui3
+cui4
+cun1
+cun2
+cun4
+cuo1
+cuo2
+cuo4
+d
+da
+da1
+da2
+da3
+da4
+dai1
+dai2
+dai3
+dai4
+dan1
+dan2
+dan3
+dan4
+dang1
+dang2
+dang3
+dang4
+dao1
+dao2
+dao3
+dao4
+de
+de1
+de2
+dei3
+den4
+deng1
+deng2
+deng3
+deng4
+di1
+di2
+di3
+di4
+dia3
+dian1
+dian2
+dian3
+dian4
+diao1
+diao3
+diao4
+die1
+die2
+die4
+ding1
+ding2
+ding3
+ding4
+diu1
+dong1
+dong3
+dong4
+dou1
+dou2
+dou3
+dou4
+du1
+du2
+du3
+du4
+duan1
+duan2
+duan3
+duan4
+dui1
+dui4
+dun1
+dun3
+dun4
+duo1
+duo2
+duo3
+duo4
+e
+e1
+e2
+e3
+e4
+ei2
+en1
+en4
+er
+er2
+er3
+er4
+f
+fa1
+fa2
+fa3
+fa4
+fan1
+fan2
+fan3
+fan4
+fang1
+fang2
+fang3
+fang4
+fei1
+fei2
+fei3
+fei4
+fen1
+fen2
+fen3
+fen4
+feng1
+feng2
+feng3
+feng4
+fo2
+fou2
+fou3
+fu1
+fu2
+fu3
+fu4
+g
+ga1
+ga2
+ga3
+ga4
+gai1
+gai2
+gai3
+gai4
+gan1
+gan2
+gan3
+gan4
+gang1
+gang2
+gang3
+gang4
+gao1
+gao2
+gao3
+gao4
+ge1
+ge2
+ge3
+ge4
+gei2
+gei3
+gen1
+gen2
+gen3
+gen4
+geng1
+geng3
+geng4
+gong1
+gong3
+gong4
+gou1
+gou2
+gou3
+gou4
+gu
+gu1
+gu2
+gu3
+gu4
+gua1
+gua2
+gua3
+gua4
+guai1
+guai2
+guai3
+guai4
+guan1
+guan2
+guan3
+guan4
+guang1
+guang2
+guang3
+guang4
+gui1
+gui2
+gui3
+gui4
+gun3
+gun4
+guo1
+guo2
+guo3
+guo4
+h
+ha1
+ha2
+ha3
+hai1
+hai2
+hai3
+hai4
+han1
+han2
+han3
+han4
+hang1
+hang2
+hang4
+hao1
+hao2
+hao3
+hao4
+he1
+he2
+he4
+hei1
+hen2
+hen3
+hen4
+heng1
+heng2
+heng4
+hong1
+hong2
+hong3
+hong4
+hou1
+hou2
+hou3
+hou4
+hu1
+hu2
+hu3
+hu4
+hua1
+hua2
+hua4
+huai2
+huai4
+huan1
+huan2
+huan3
+huan4
+huang1
+huang2
+huang3
+huang4
+hui1
+hui2
+hui3
+hui4
+hun1
+hun2
+hun4
+huo
+huo1
+huo2
+huo3
+huo4
+i
+j
+ji1
+ji2
+ji3
+ji4
+jia
+jia1
+jia2
+jia3
+jia4
+jian1
+jian2
+jian3
+jian4
+jiang1
+jiang2
+jiang3
+jiang4
+jiao1
+jiao2
+jiao3
+jiao4
+jie1
+jie2
+jie3
+jie4
+jin1
+jin2
+jin3
+jin4
+jing1
+jing2
+jing3
+jing4
+jiong3
+jiu1
+jiu2
+jiu3
+jiu4
+ju1
+ju2
+ju3
+ju4
+juan1
+juan2
+juan3
+juan4
+jue1
+jue2
+jue4
+jun1
+jun4
+k
+ka1
+ka2
+ka3
+kai1
+kai2
+kai3
+kai4
+kan1
+kan2
+kan3
+kan4
+kang1
+kang2
+kang4
+kao1
+kao2
+kao3
+kao4
+ke1
+ke2
+ke3
+ke4
+ken3
+keng1
+kong1
+kong3
+kong4
+kou1
+kou2
+kou3
+kou4
+ku1
+ku2
+ku3
+ku4
+kua1
+kua3
+kua4
+kuai3
+kuai4
+kuan1
+kuan2
+kuan3
+kuang1
+kuang2
+kuang4
+kui1
+kui2
+kui3
+kui4
+kun1
+kun3
+kun4
+kuo4
+l
+la
+la1
+la2
+la3
+la4
+lai2
+lai4
+lan2
+lan3
+lan4
+lang1
+lang2
+lang3
+lang4
+lao1
+lao2
+lao3
+lao4
+le
+le1
+le4
+lei
+lei1
+lei2
+lei3
+lei4
+leng1
+leng2
+leng3
+leng4
+li
+li1
+li2
+li3
+li4
+lia3
+lian2
+lian3
+lian4
+liang2
+liang3
+liang4
+liao1
+liao2
+liao3
+liao4
+lie1
+lie2
+lie3
+lie4
+lin1
+lin2
+lin3
+lin4
+ling2
+ling3
+ling4
+liu1
+liu2
+liu3
+liu4
+long1
+long2
+long3
+long4
+lou1
+lou2
+lou3
+lou4
+lu1
+lu2
+lu3
+lu4
+luan2
+luan3
+luan4
+lun1
+lun2
+lun4
+luo1
+luo2
+luo3
+luo4
+lv2
+lv3
+lv4
+lve3
+lve4
+m
+ma
+ma1
+ma2
+ma3
+ma4
+mai2
+mai3
+mai4
+man1
+man2
+man3
+man4
+mang2
+mang3
+mao1
+mao2
+mao3
+mao4
+me
+mei2
+mei3
+mei4
+men
+men1
+men2
+men4
+meng
+meng1
+meng2
+meng3
+meng4
+mi1
+mi2
+mi3
+mi4
+mian2
+mian3
+mian4
+miao1
+miao2
+miao3
+miao4
+mie1
+mie4
+min2
+min3
+ming2
+ming3
+ming4
+miu4
+mo1
+mo2
+mo3
+mo4
+mou1
+mou2
+mou3
+mu2
+mu3
+mu4
+n
+n2
+na1
+na2
+na3
+na4
+nai2
+nai3
+nai4
+nan1
+nan2
+nan3
+nan4
+nang1
+nang2
+nang3
+nao1
+nao2
+nao3
+nao4
+ne
+ne2
+ne4
+nei3
+nei4
+nen4
+neng2
+ni1
+ni2
+ni3
+ni4
+nian1
+nian2
+nian3
+nian4
+niang2
+niang4
+niao2
+niao3
+niao4
+nie1
+nie4
+nin2
+ning2
+ning3
+ning4
+niu1
+niu2
+niu3
+niu4
+nong2
+nong4
+nou4
+nu2
+nu3
+nu4
+nuan3
+nuo2
+nuo4
+nv2
+nv3
+nve4
+o
+o1
+o2
+ou1
+ou2
+ou3
+ou4
+p
+pa1
+pa2
+pa4
+pai1
+pai2
+pai3
+pai4
+pan1
+pan2
+pan4
+pang1
+pang2
+pang4
+pao1
+pao2
+pao3
+pao4
+pei1
+pei2
+pei4
+pen1
+pen2
+pen4
+peng1
+peng2
+peng3
+peng4
+pi1
+pi2
+pi3
+pi4
+pian1
+pian2
+pian4
+piao1
+piao2
+piao3
+piao4
+pie1
+pie2
+pie3
+pin1
+pin2
+pin3
+pin4
+ping1
+ping2
+po1
+po2
+po3
+po4
+pou1
+pu1
+pu2
+pu3
+pu4
+q
+qi1
+qi2
+qi3
+qi4
+qia1
+qia3
+qia4
+qian1
+qian2
+qian3
+qian4
+qiang1
+qiang2
+qiang3
+qiang4
+qiao1
+qiao2
+qiao3
+qiao4
+qie1
+qie2
+qie3
+qie4
+qin1
+qin2
+qin3
+qin4
+qing1
+qing2
+qing3
+qing4
+qiong1
+qiong2
+qiu1
+qiu2
+qiu3
+qu1
+qu2
+qu3
+qu4
+quan1
+quan2
+quan3
+quan4
+que1
+que2
+que4
+qun2
+r
+ran2
+ran3
+rang1
+rang2
+rang3
+rang4
+rao2
+rao3
+rao4
+re2
+re3
+re4
+ren2
+ren3
+ren4
+reng1
+reng2
+ri4
+rong1
+rong2
+rong3
+rou2
+rou4
+ru2
+ru3
+ru4
+ruan2
+ruan3
+rui3
+rui4
+run4
+ruo4
+s
+sa1
+sa2
+sa3
+sa4
+sai1
+sai4
+san1
+san2
+san3
+san4
+sang1
+sang3
+sang4
+sao1
+sao2
+sao3
+sao4
+se4
+sen1
+seng1
+sha1
+sha2
+sha3
+sha4
+shai1
+shai2
+shai3
+shai4
+shan1
+shan3
+shan4
+shang
+shang1
+shang3
+shang4
+shao1
+shao2
+shao3
+shao4
+she1
+she2
+she3
+she4
+shei2
+shen1
+shen2
+shen3
+shen4
+sheng1
+sheng2
+sheng3
+sheng4
+shi
+shi1
+shi2
+shi3
+shi4
+shou1
+shou2
+shou3
+shou4
+shu1
+shu2
+shu3
+shu4
+shua1
+shua2
+shua3
+shua4
+shuai1
+shuai3
+shuai4
+shuan1
+shuan4
+shuang1
+shuang3
+shui2
+shui3
+shui4
+shun3
+shun4
+shuo1
+shuo4
+si1
+si2
+si3
+si4
+song1
+song3
+song4
+sou1
+sou3
+sou4
+su1
+su2
+su4
+suan1
+suan4
+sui1
+sui2
+sui3
+sui4
+sun1
+sun3
+suo
+suo1
+suo2
+suo3
+t
+ta1
+ta2
+ta3
+ta4
+tai1
+tai2
+tai4
+tan1
+tan2
+tan3
+tan4
+tang1
+tang2
+tang3
+tang4
+tao1
+tao2
+tao3
+tao4
+te4
+teng2
+ti1
+ti2
+ti3
+ti4
+tian1
+tian2
+tian3
+tiao1
+tiao2
+tiao3
+tiao4
+tie1
+tie2
+tie3
+tie4
+ting1
+ting2
+ting3
+tong1
+tong2
+tong3
+tong4
+tou
+tou1
+tou2
+tou4
+tu1
+tu2
+tu3
+tu4
+tuan1
+tuan2
+tui1
+tui2
+tui3
+tui4
+tun1
+tun2
+tun4
+tuo1
+tuo2
+tuo3
+tuo4
+u
+v
+w
+wa
+wa1
+wa2
+wa3
+wa4
+wai1
+wai3
+wai4
+wan1
+wan2
+wan3
+wan4
+wang1
+wang2
+wang3
+wang4
+wei1
+wei2
+wei3
+wei4
+wen1
+wen2
+wen3
+wen4
+weng1
+weng4
+wo1
+wo2
+wo3
+wo4
+wu1
+wu2
+wu3
+wu4
+x
+xi1
+xi2
+xi3
+xi4
+xia1
+xia2
+xia4
+xian1
+xian2
+xian3
+xian4
+xiang1
+xiang2
+xiang3
+xiang4
+xiao1
+xiao2
+xiao3
+xiao4
+xie1
+xie2
+xie3
+xie4
+xin1
+xin2
+xin4
+xing1
+xing2
+xing3
+xing4
+xiong1
+xiong2
+xiu1
+xiu3
+xiu4
+xu
+xu1
+xu2
+xu3
+xu4
+xuan1
+xuan2
+xuan3
+xuan4
+xue1
+xue2
+xue3
+xue4
+xun1
+xun2
+xun4
+y
+ya
+ya1
+ya2
+ya3
+ya4
+yan1
+yan2
+yan3
+yan4
+yang1
+yang2
+yang3
+yang4
+yao1
+yao2
+yao3
+yao4
+ye1
+ye2
+ye3
+ye4
+yi
+yi1
+yi2
+yi3
+yi4
+yin1
+yin2
+yin3
+yin4
+ying1
+ying2
+ying3
+ying4
+yo1
+yong1
+yong2
+yong3
+yong4
+you1
+you2
+you3
+you4
+yu1
+yu2
+yu3
+yu4
+yuan1
+yuan2
+yuan3
+yuan4
+yue1
+yue4
+yun1
+yun2
+yun3
+yun4
+z
+za1
+za2
+za3
+zai1
+zai3
+zai4
+zan1
+zan2
+zan3
+zan4
+zang1
+zang4
+zao1
+zao2
+zao3
+zao4
+ze2
+ze4
+zei2
+zen3
+zeng1
+zeng4
+zha1
+zha2
+zha3
+zha4
+zhai1
+zhai2
+zhai3
+zhai4
+zhan1
+zhan2
+zhan3
+zhan4
+zhang1
+zhang2
+zhang3
+zhang4
+zhao1
+zhao2
+zhao3
+zhao4
+zhe
+zhe1
+zhe2
+zhe3
+zhe4
+zhen1
+zhen2
+zhen3
+zhen4
+zheng1
+zheng2
+zheng3
+zheng4
+zhi1
+zhi2
+zhi3
+zhi4
+zhong1
+zhong2
+zhong3
+zhong4
+zhou1
+zhou2
+zhou3
+zhou4
+zhu1
+zhu2
+zhu3
+zhu4
+zhua1
+zhua2
+zhua3
+zhuai1
+zhuai3
+zhuai4
+zhuan1
+zhuan2
+zhuan3
+zhuan4
+zhuang1
+zhuang4
+zhui1
+zhui4
+zhun1
+zhun2
+zhun3
+zhuo1
+zhuo2
+zi
+zi1
+zi2
+zi3
+zi4
+zong1
+zong2
+zong3
+zong4
+zou1
+zou2
+zou3
+zou4
+zu1
+zu2
+zu3
+zuan1
+zuan3
+zuan4
+zui2
+zui3
+zui4
+zun1
+zuo
+zuo1
+zuo2
+zuo3
+zuo4
+{
+~
+¡
+¢
+£
+¥
+§
+¨
+©
+«
+®
+¯
+°
+±
+²
+³
+´
+µ
+·
+¹
+º
+»
+¼
+½
+¾
+¿
+À
+Á
+Â
+Ã
+Ä
+Å
+Æ
+Ç
+È
+É
+Ê
+Í
+Î
+Ñ
+Ó
+Ö
+×
+Ø
+Ú
+Ü
+Ý
+Þ
+ß
+à
+á
+â
+ã
+ä
+å
+æ
+ç
+è
+é
+ê
+ë
+ì
+í
+î
+ï
+ð
+ñ
+ò
+ó
+ô
+õ
+ö
+ø
+ù
+ú
+û
+ü
+ý
+Ā
+ā
+ă
+ą
+ć
+Č
+č
+Đ
+đ
+ē
+ė
+ę
+ě
+ĝ
+ğ
+ħ
+ī
+į
+İ
+ı
+Ł
+ł
+ń
+ņ
+ň
+ŋ
+Ō
+ō
+ő
+œ
+ř
+Ś
+ś
+Ş
+ş
+Š
+š
+Ť
+ť
+ũ
+ū
+ź
+Ż
+ż
+Ž
+ž
+ơ
+ư
+ǎ
+ǐ
+ǒ
+ǔ
+ǚ
+ș
+ț
+ɑ
+ɔ
+ɕ
+ə
+ɛ
+ɜ
+ɡ
+ɣ
+ɪ
+ɫ
+ɴ
+ɹ
+ɾ
+ʃ
+ʊ
+ʌ
+ʒ
+ʔ
+ʰ
+ʷ
+ʻ
+ʾ
+ʿ
+ˈ
+ː
+˙
+˜
+ˢ
+́
+̅
+Α
+Β
+Δ
+Ε
+Θ
+Κ
+Λ
+Μ
+Ξ
+Π
+Σ
+Τ
+Φ
+Χ
+Ψ
+Ω
+ά
+έ
+ή
+ί
+α
+β
+γ
+δ
+ε
+ζ
+η
+θ
+ι
+κ
+λ
+μ
+ν
+ξ
+ο
+π
+ρ
+ς
+σ
+τ
+υ
+φ
+χ
+ψ
+ω
+ϊ
+ό
+ύ
+ώ
+ϕ
+ϵ
+Ё
+А
+Б
+В
+Г
+Д
+Е
+Ж
+З
+И
+Й
+К
+Л
+М
+Н
+О
+П
+Р
+С
+Т
+У
+Ф
+Х
+Ц
+Ч
+Ш
+Щ
+Ы
+Ь
+Э
+Ю
+Я
+а
+б
+в
+г
+д
+е
+ж
+з
+и
+й
+к
+л
+м
+н
+о
+п
+р
+с
+т
+у
+ф
+х
+ц
+ч
+ш
+щ
+ъ
+ы
+ь
+э
+ю
+я
+ё
+і
+ְ
+ִ
+ֵ
+ֶ
+ַ
+ָ
+ֹ
+ּ
+־
+ׁ
+א
+ב
+ג
+ד
+ה
+ו
+ז
+ח
+ט
+י
+כ
+ל
+ם
+מ
+ן
+נ
+ס
+ע
+פ
+ק
+ר
+ש
+ת
+أ
+ب
+ة
+ت
+ج
+ح
+د
+ر
+ز
+س
+ص
+ط
+ع
+ق
+ك
+ل
+م
+ن
+ه
+و
+ي
+َ
+ُ
+ِ
+ْ
+ก
+ข
+ง
+จ
+ต
+ท
+น
+ป
+ย
+ร
+ว
+ส
+ห
+อ
+ฮ
+ั
+า
+ี
+ึ
+โ
+ใ
+ไ
+่
+้
+์
+ḍ
+Ḥ
+ḥ
+ṁ
+ṃ
+ṅ
+ṇ
+Ṛ
+ṛ
+Ṣ
+ṣ
+Ṭ
+ṭ
+ạ
+ả
+Ấ
+ấ
+ầ
+ậ
+ắ
+ằ
+ẻ
+ẽ
+ế
+ề
+ể
+ễ
+ệ
+ị
+ọ
+ỏ
+ố
+ồ
+ộ
+ớ
+ờ
+ở
+ụ
+ủ
+ứ
+ữ
+ἀ
+ἁ
+Ἀ
+ἐ
+ἔ
+ἰ
+ἱ
+ὀ
+ὁ
+ὐ
+ὲ
+ὸ
+ᾶ
+᾽
+ῆ
+ῇ
+ῶ
+‎
+‑
+‒
+–
+—
+―
+‖
+†
+‡
+•
+…
+‧
+‬
+′
+″
+⁄
+⁡
+⁰
+⁴
+⁵
+⁶
+⁷
+⁸
+⁹
+₁
+₂
+₃
+€
+₱
+₹
+₽
+℃
+ℏ
+ℓ
+№
+ℝ
+™
+⅓
+⅔
+⅛
+→
+∂
+∈
+∑
+−
+∗
+√
+∞
+∫
+≈
+≠
+≡
+≤
+≥
+⋅
+⋯
+█
+♪
+⟨
+⟩
+、
+。
+《
+》
+「
+」
+【
+】
+あ
+う
+え
+お
+か
+が
+き
+ぎ
+く
+ぐ
+け
+げ
+こ
+ご
+さ
+し
+じ
+す
+ず
+せ
+ぜ
+そ
+ぞ
+た
+だ
+ち
+っ
+つ
+で
+と
+ど
+な
+に
+ね
+の
+は
+ば
+ひ
+ぶ
+へ
+べ
+ま
+み
+む
+め
+も
+ゃ
+や
+ゆ
+ょ
+よ
+ら
+り
+る
+れ
+ろ
+わ
+を
+ん
+ァ
+ア
+ィ
+イ
+ウ
+ェ
+エ
+オ
+カ
+ガ
+キ
+ク
+ケ
+ゲ
+コ
+ゴ
+サ
+ザ
+シ
+ジ
+ス
+ズ
+セ
+ゾ
+タ
+ダ
+チ
+ッ
+ツ
+テ
+デ
+ト
+ド
+ナ
+ニ
+ネ
+ノ
+バ
+パ
+ビ
+ピ
+フ
+プ
+ヘ
+ベ
+ペ
+ホ
+ボ
+ポ
+マ
+ミ
+ム
+メ
+モ
+ャ
+ヤ
+ュ
+ユ
+ョ
+ヨ
+ラ
+リ
+ル
+レ
+ロ
+ワ
+ン
+・
+ー
+ㄋ
+ㄍ
+ㄎ
+ㄏ
+ㄓ
+ㄕ
+ㄚ
+ㄜ
+ㄟ
+ㄤ
+ㄥ
+ㄧ
+ㄱ
+ㄴ
+ㄷ
+ㄹ
+ㅁ
+ㅂ
+ㅅ
+ㅈ
+ㅍ
+ㅎ
+ㅏ
+ㅓ
+ㅗ
+ㅜ
+ㅡ
+ㅣ
+㗎
+가
+각
+간
+갈
+감
+갑
+갓
+갔
+강
+같
+개
+거
+건
+걸
+겁
+것
+겉
+게
+겠
+겨
+결
+겼
+경
+계
+고
+곤
+골
+곱
+공
+과
+관
+광
+교
+구
+국
+굴
+귀
+귄
+그
+근
+글
+금
+기
+긴
+길
+까
+깍
+깔
+깜
+깨
+께
+꼬
+꼭
+꽃
+꾸
+꿔
+끔
+끗
+끝
+끼
+나
+난
+날
+남
+납
+내
+냐
+냥
+너
+넘
+넣
+네
+녁
+년
+녕
+노
+녹
+놀
+누
+눈
+느
+는
+늘
+니
+님
+닙
+다
+닥
+단
+달
+닭
+당
+대
+더
+덕
+던
+덥
+데
+도
+독
+동
+돼
+됐
+되
+된
+될
+두
+둑
+둥
+드
+들
+등
+디
+따
+딱
+딸
+땅
+때
+떤
+떨
+떻
+또
+똑
+뚱
+뛰
+뜻
+띠
+라
+락
+란
+람
+랍
+랑
+래
+랜
+러
+런
+럼
+렇
+레
+려
+력
+렵
+렸
+로
+록
+롬
+루
+르
+른
+를
+름
+릉
+리
+릴
+림
+마
+막
+만
+많
+말
+맑
+맙
+맛
+매
+머
+먹
+멍
+메
+면
+명
+몇
+모
+목
+몸
+못
+무
+문
+물
+뭐
+뭘
+미
+민
+밌
+밑
+바
+박
+밖
+반
+받
+발
+밤
+밥
+방
+배
+백
+밸
+뱀
+버
+번
+벌
+벚
+베
+벼
+벽
+별
+병
+보
+복
+본
+볼
+봐
+봤
+부
+분
+불
+비
+빔
+빛
+빠
+빨
+뼈
+뽀
+뿅
+쁘
+사
+산
+살
+삼
+샀
+상
+새
+색
+생
+서
+선
+설
+섭
+섰
+성
+세
+셔
+션
+셨
+소
+속
+손
+송
+수
+숙
+순
+술
+숫
+숭
+숲
+쉬
+쉽
+스
+슨
+습
+슷
+시
+식
+신
+실
+싫
+심
+십
+싶
+싸
+써
+쓰
+쓴
+씌
+씨
+씩
+씬
+아
+악
+안
+않
+알
+야
+약
+얀
+양
+얘
+어
+언
+얼
+엄
+업
+없
+었
+엉
+에
+여
+역
+연
+염
+엽
+영
+옆
+예
+옛
+오
+온
+올
+옷
+옹
+와
+왔
+왜
+요
+욕
+용
+우
+운
+울
+웃
+워
+원
+월
+웠
+위
+윙
+유
+육
+윤
+으
+은
+을
+음
+응
+의
+이
+익
+인
+일
+읽
+임
+입
+있
+자
+작
+잔
+잖
+잘
+잡
+잤
+장
+재
+저
+전
+점
+정
+제
+져
+졌
+조
+족
+좀
+종
+좋
+죠
+주
+준
+줄
+중
+줘
+즈
+즐
+즘
+지
+진
+집
+짜
+짝
+쩌
+쪼
+쪽
+쫌
+쭈
+쯔
+찌
+찍
+차
+착
+찾
+책
+처
+천
+철
+체
+쳐
+쳤
+초
+촌
+추
+출
+춤
+춥
+춰
+치
+친
+칠
+침
+칩
+칼
+커
+켓
+코
+콩
+쿠
+퀴
+크
+큰
+큽
+키
+킨
+타
+태
+터
+턴
+털
+테
+토
+통
+투
+트
+특
+튼
+틀
+티
+팀
+파
+팔
+패
+페
+펜
+펭
+평
+포
+폭
+표
+품
+풍
+프
+플
+피
+필
+하
+학
+한
+할
+함
+합
+항
+해
+햇
+했
+행
+허
+험
+형
+혜
+호
+혼
+홀
+화
+회
+획
+후
+휴
+흐
+흔
+희
+히
+힘
+ﷺ
+ﷻ
+！
+，
+？
+�
+𠮶

2flow/models/downloads/F5TTS_Base_bigvgan/model_1250000.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bdab3e92fc2b77447aa8c46aac77531d970822b191ca198e5ab94aef99265df9
+size 1348555394

2flow/models/downloads/F5TTS_v1_Base/model_1250000.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:670900fd14e6c458b95da6e9ed317cdb20dbaf7a1c02ac06a05475a9d32b6a38
+size 1348435761

2flow/models/downloads/F5TTS_v1_Base/vocab.txt ADDED Viewed

	@@ -0,0 +1,2545 @@

+!
+"
+#
+$
+%
+&
+'
+(
+)
+*
++
+,
+-
+.
+/
+0
+1
+2
+3
+4
+5
+6
+7
+8
+9
+:
+;
+=
+>
+?
+@
+A
+B
+C
+D
+E
+F
+G
+H
+I
+J
+K
+L
+M
+N
+O
+P
+Q
+R
+S
+T
+U
+V
+W
+X
+Y
+Z
+[
+\
+]
+_
+a
+a1
+ai1
+ai2
+ai3
+ai4
+an1
+an3
+an4
+ang1
+ang2
+ang4
+ao1
+ao2
+ao3
+ao4
+b
+ba
+ba1
+ba2
+ba3
+ba4
+bai1
+bai2
+bai3
+bai4
+ban1
+ban2
+ban3
+ban4
+bang1
+bang2
+bang3
+bang4
+bao1
+bao2
+bao3
+bao4
+bei
+bei1
+bei2
+bei3
+bei4
+ben1
+ben2
+ben3
+ben4
+beng
+beng1
+beng2
+beng3
+beng4
+bi1
+bi2
+bi3
+bi4
+bian1
+bian2
+bian3
+bian4
+biao1
+biao2
+biao3
+bie1
+bie2
+bie3
+bie4
+bin1
+bin4
+bing1
+bing2
+bing3
+bing4
+bo
+bo1
+bo2
+bo3
+bo4
+bu2
+bu3
+bu4
+c
+ca1
+cai1
+cai2
+cai3
+cai4
+can1
+can2
+can3
+can4
+cang1
+cang2
+cao1
+cao2
+cao3
+ce4
+cen1
+cen2
+ceng1
+ceng2
+ceng4
+cha1
+cha2
+cha3
+cha4
+chai1
+chai2
+chan1
+chan2
+chan3
+chan4
+chang1
+chang2
+chang3
+chang4
+chao1
+chao2
+chao3
+che1
+che2
+che3
+che4
+chen1
+chen2
+chen3
+chen4
+cheng1
+cheng2
+cheng3
+cheng4
+chi1
+chi2
+chi3
+chi4
+chong1
+chong2
+chong3
+chong4
+chou1
+chou2
+chou3
+chou4
+chu1
+chu2
+chu3
+chu4
+chua1
+chuai1
+chuai2
+chuai3
+chuai4
+chuan1
+chuan2
+chuan3
+chuan4
+chuang1
+chuang2
+chuang3
+chuang4
+chui1
+chui2
+chun1
+chun2
+chun3
+chuo1
+chuo4
+ci1
+ci2
+ci3
+ci4
+cong1
+cong2
+cou4
+cu1
+cu4
+cuan1
+cuan2
+cuan4
+cui1
+cui3
+cui4
+cun1
+cun2
+cun4
+cuo1
+cuo2
+cuo4
+d
+da
+da1
+da2
+da3
+da4
+dai1
+dai2
+dai3
+dai4
+dan1
+dan2
+dan3
+dan4
+dang1
+dang2
+dang3
+dang4
+dao1
+dao2
+dao3
+dao4
+de
+de1
+de2
+dei3
+den4
+deng1
+deng2
+deng3
+deng4
+di1
+di2
+di3
+di4
+dia3
+dian1
+dian2
+dian3
+dian4
+diao1
+diao3
+diao4
+die1
+die2
+die4
+ding1
+ding2
+ding3
+ding4
+diu1
+dong1
+dong3
+dong4
+dou1
+dou2
+dou3
+dou4
+du1
+du2
+du3
+du4
+duan1
+duan2
+duan3
+duan4
+dui1
+dui4
+dun1
+dun3
+dun4
+duo1
+duo2
+duo3
+duo4
+e
+e1
+e2
+e3
+e4
+ei2
+en1
+en4
+er
+er2
+er3
+er4
+f
+fa1
+fa2
+fa3
+fa4
+fan1
+fan2
+fan3
+fan4
+fang1
+fang2
+fang3
+fang4
+fei1
+fei2
+fei3
+fei4
+fen1
+fen2
+fen3
+fen4
+feng1
+feng2
+feng3
+feng4
+fo2
+fou2
+fou3
+fu1
+fu2
+fu3
+fu4
+g
+ga1
+ga2
+ga3
+ga4
+gai1
+gai2
+gai3
+gai4
+gan1
+gan2
+gan3
+gan4
+gang1
+gang2
+gang3
+gang4
+gao1
+gao2
+gao3
+gao4
+ge1
+ge2
+ge3
+ge4
+gei2
+gei3
+gen1
+gen2
+gen3
+gen4
+geng1
+geng3
+geng4
+gong1
+gong3
+gong4
+gou1
+gou2
+gou3
+gou4
+gu
+gu1
+gu2
+gu3
+gu4
+gua1
+gua2
+gua3
+gua4
+guai1
+guai2
+guai3
+guai4
+guan1
+guan2
+guan3
+guan4
+guang1
+guang2
+guang3
+guang4
+gui1
+gui2
+gui3
+gui4
+gun3
+gun4
+guo1
+guo2
+guo3
+guo4
+h
+ha1
+ha2
+ha3
+hai1
+hai2
+hai3
+hai4
+han1
+han2
+han3
+han4
+hang1
+hang2
+hang4
+hao1
+hao2
+hao3
+hao4
+he1
+he2
+he4
+hei1
+hen2
+hen3
+hen4
+heng1
+heng2
+heng4
+hong1
+hong2
+hong3
+hong4
+hou1
+hou2
+hou3
+hou4
+hu1
+hu2
+hu3
+hu4
+hua1
+hua2
+hua4
+huai2
+huai4
+huan1
+huan2
+huan3
+huan4
+huang1
+huang2
+huang3
+huang4
+hui1
+hui2
+hui3
+hui4
+hun1
+hun2
+hun4
+huo
+huo1
+huo2
+huo3
+huo4
+i
+j
+ji1
+ji2
+ji3
+ji4
+jia
+jia1
+jia2
+jia3
+jia4
+jian1
+jian2
+jian3
+jian4
+jiang1
+jiang2
+jiang3
+jiang4
+jiao1
+jiao2
+jiao3
+jiao4
+jie1
+jie2
+jie3
+jie4
+jin1
+jin2
+jin3
+jin4
+jing1
+jing2
+jing3
+jing4
+jiong3
+jiu1
+jiu2
+jiu3
+jiu4
+ju1
+ju2
+ju3
+ju4
+juan1
+juan2
+juan3
+juan4
+jue1
+jue2
+jue4
+jun1
+jun4
+k
+ka1
+ka2
+ka3
+kai1
+kai2
+kai3
+kai4
+kan1
+kan2
+kan3
+kan4
+kang1
+kang2
+kang4
+kao1
+kao2
+kao3
+kao4
+ke1
+ke2
+ke3
+ke4
+ken3
+keng1
+kong1
+kong3
+kong4
+kou1
+kou2
+kou3
+kou4
+ku1
+ku2
+ku3
+ku4
+kua1
+kua3
+kua4
+kuai3
+kuai4
+kuan1
+kuan2
+kuan3
+kuang1
+kuang2
+kuang4
+kui1
+kui2
+kui3
+kui4
+kun1
+kun3
+kun4
+kuo4
+l
+la
+la1
+la2
+la3
+la4
+lai2
+lai4
+lan2
+lan3
+lan4
+lang1
+lang2
+lang3
+lang4
+lao1
+lao2
+lao3
+lao4
+le
+le1
+le4
+lei
+lei1
+lei2
+lei3
+lei4
+leng1
+leng2
+leng3
+leng4
+li
+li1
+li2
+li3
+li4
+lia3
+lian2
+lian3
+lian4
+liang2
+liang3
+liang4
+liao1
+liao2
+liao3
+liao4
+lie1
+lie2
+lie3
+lie4
+lin1
+lin2
+lin3
+lin4
+ling2
+ling3
+ling4
+liu1
+liu2
+liu3
+liu4
+long1
+long2
+long3
+long4
+lou1
+lou2
+lou3
+lou4
+lu1
+lu2
+lu3
+lu4
+luan2
+luan3
+luan4
+lun1
+lun2
+lun4
+luo1
+luo2
+luo3
+luo4
+lv2
+lv3
+lv4
+lve3
+lve4
+m
+ma
+ma1
+ma2
+ma3
+ma4
+mai2
+mai3
+mai4
+man1
+man2
+man3
+man4
+mang2
+mang3
+mao1
+mao2
+mao3
+mao4
+me
+mei2
+mei3
+mei4
+men
+men1
+men2
+men4
+meng
+meng1
+meng2
+meng3
+meng4
+mi1
+mi2
+mi3
+mi4
+mian2
+mian3
+mian4
+miao1
+miao2
+miao3
+miao4
+mie1
+mie4
+min2
+min3
+ming2
+ming3
+ming4
+miu4
+mo1
+mo2
+mo3
+mo4
+mou1
+mou2
+mou3
+mu2
+mu3
+mu4
+n
+n2
+na1
+na2
+na3
+na4
+nai2
+nai3
+nai4
+nan1
+nan2
+nan3
+nan4
+nang1
+nang2
+nang3
+nao1
+nao2
+nao3
+nao4
+ne
+ne2
+ne4
+nei3
+nei4
+nen4
+neng2
+ni1
+ni2
+ni3
+ni4
+nian1
+nian2
+nian3
+nian4
+niang2
+niang4
+niao2
+niao3
+niao4
+nie1
+nie4
+nin2
+ning2
+ning3
+ning4
+niu1
+niu2
+niu3
+niu4
+nong2
+nong4
+nou4
+nu2
+nu3
+nu4
+nuan3
+nuo2
+nuo4
+nv2
+nv3
+nve4
+o
+o1
+o2
+ou1
+ou2
+ou3
+ou4
+p
+pa1
+pa2
+pa4
+pai1
+pai2
+pai3
+pai4
+pan1
+pan2
+pan4
+pang1
+pang2
+pang4
+pao1
+pao2
+pao3
+pao4
+pei1
+pei2
+pei4
+pen1
+pen2
+pen4
+peng1
+peng2
+peng3
+peng4
+pi1
+pi2
+pi3
+pi4
+pian1
+pian2
+pian4
+piao1
+piao2
+piao3
+piao4
+pie1
+pie2
+pie3
+pin1
+pin2
+pin3
+pin4
+ping1
+ping2
+po1
+po2
+po3
+po4
+pou1
+pu1
+pu2
+pu3
+pu4
+q
+qi1
+qi2
+qi3
+qi4
+qia1
+qia3
+qia4
+qian1
+qian2
+qian3
+qian4
+qiang1
+qiang2
+qiang3
+qiang4
+qiao1
+qiao2
+qiao3
+qiao4
+qie1
+qie2
+qie3
+qie4
+qin1
+qin2
+qin3
+qin4
+qing1
+qing2
+qing3
+qing4
+qiong1
+qiong2
+qiu1
+qiu2
+qiu3
+qu1
+qu2
+qu3
+qu4
+quan1
+quan2
+quan3
+quan4
+que1
+que2
+que4
+qun2
+r
+ran2
+ran3
+rang1
+rang2
+rang3
+rang4
+rao2
+rao3
+rao4
+re2
+re3
+re4
+ren2
+ren3
+ren4
+reng1
+reng2
+ri4
+rong1
+rong2
+rong3
+rou2
+rou4
+ru2
+ru3
+ru4
+ruan2
+ruan3
+rui3
+rui4
+run4
+ruo4
+s
+sa1
+sa2
+sa3
+sa4
+sai1
+sai4
+san1
+san2
+san3
+san4
+sang1
+sang3
+sang4
+sao1
+sao2
+sao3
+sao4
+se4
+sen1
+seng1
+sha1
+sha2
+sha3
+sha4
+shai1
+shai2
+shai3
+shai4
+shan1
+shan3
+shan4
+shang
+shang1
+shang3
+shang4
+shao1
+shao2
+shao3
+shao4
+she1
+she2
+she3
+she4
+shei2
+shen1
+shen2
+shen3
+shen4
+sheng1
+sheng2
+sheng3
+sheng4
+shi
+shi1
+shi2
+shi3
+shi4
+shou1
+shou2
+shou3
+shou4
+shu1
+shu2
+shu3
+shu4
+shua1
+shua2
+shua3
+shua4
+shuai1
+shuai3
+shuai4
+shuan1
+shuan4
+shuang1
+shuang3
+shui2
+shui3
+shui4
+shun3
+shun4
+shuo1
+shuo4
+si1
+si2
+si3
+si4
+song1
+song3
+song4
+sou1
+sou3
+sou4
+su1
+su2
+su4
+suan1
+suan4
+sui1
+sui2
+sui3
+sui4
+sun1
+sun3
+suo
+suo1
+suo2
+suo3
+t
+ta1
+ta2
+ta3
+ta4
+tai1
+tai2
+tai4
+tan1
+tan2
+tan3
+tan4
+tang1
+tang2
+tang3
+tang4
+tao1
+tao2
+tao3
+tao4
+te4
+teng2
+ti1
+ti2
+ti3
+ti4
+tian1
+tian2
+tian3
+tiao1
+tiao2
+tiao3
+tiao4
+tie1
+tie2
+tie3
+tie4
+ting1
+ting2
+ting3
+tong1
+tong2
+tong3
+tong4
+tou
+tou1
+tou2
+tou4
+tu1
+tu2
+tu3
+tu4
+tuan1
+tuan2
+tui1
+tui2
+tui3
+tui4
+tun1
+tun2
+tun4
+tuo1
+tuo2
+tuo3
+tuo4
+u
+v
+w
+wa
+wa1
+wa2
+wa3
+wa4
+wai1
+wai3
+wai4
+wan1
+wan2
+wan3
+wan4
+wang1
+wang2
+wang3
+wang4
+wei1
+wei2
+wei3
+wei4
+wen1
+wen2
+wen3
+wen4
+weng1
+weng4
+wo1
+wo2
+wo3
+wo4
+wu1
+wu2
+wu3
+wu4
+x
+xi1
+xi2
+xi3
+xi4
+xia1
+xia2
+xia4
+xian1
+xian2
+xian3
+xian4
+xiang1
+xiang2
+xiang3
+xiang4
+xiao1
+xiao2
+xiao3
+xiao4
+xie1
+xie2
+xie3
+xie4
+xin1
+xin2
+xin4
+xing1
+xing2
+xing3
+xing4
+xiong1
+xiong2
+xiu1
+xiu3
+xiu4
+xu
+xu1
+xu2
+xu3
+xu4
+xuan1
+xuan2
+xuan3
+xuan4
+xue1
+xue2
+xue3
+xue4
+xun1
+xun2
+xun4
+y
+ya
+ya1
+ya2
+ya3
+ya4
+yan1
+yan2
+yan3
+yan4
+yang1
+yang2
+yang3
+yang4
+yao1
+yao2
+yao3
+yao4
+ye1
+ye2
+ye3
+ye4
+yi
+yi1
+yi2
+yi3
+yi4
+yin1
+yin2
+yin3
+yin4
+ying1
+ying2
+ying3
+ying4
+yo1
+yong1
+yong2
+yong3
+yong4
+you1
+you2
+you3
+you4
+yu1
+yu2
+yu3
+yu4
+yuan1
+yuan2
+yuan3
+yuan4
+yue1
+yue4
+yun1
+yun2
+yun3
+yun4
+z
+za1
+za2
+za3
+zai1
+zai3
+zai4
+zan1
+zan2
+zan3
+zan4
+zang1
+zang4
+zao1
+zao2
+zao3
+zao4
+ze2
+ze4
+zei2
+zen3
+zeng1
+zeng4
+zha1
+zha2
+zha3
+zha4
+zhai1
+zhai2
+zhai3
+zhai4
+zhan1
+zhan2
+zhan3
+zhan4
+zhang1
+zhang2
+zhang3
+zhang4
+zhao1
+zhao2
+zhao3
+zhao4
+zhe
+zhe1
+zhe2
+zhe3
+zhe4
+zhen1
+zhen2
+zhen3
+zhen4
+zheng1
+zheng2
+zheng3
+zheng4
+zhi1
+zhi2
+zhi3
+zhi4
+zhong1
+zhong2
+zhong3
+zhong4
+zhou1
+zhou2
+zhou3
+zhou4
+zhu1
+zhu2
+zhu3
+zhu4
+zhua1
+zhua2
+zhua3
+zhuai1
+zhuai3
+zhuai4
+zhuan1
+zhuan2
+zhuan3
+zhuan4
+zhuang1
+zhuang4
+zhui1
+zhui4
+zhun1
+zhun2
+zhun3
+zhuo1
+zhuo2
+zi
+zi1
+zi2
+zi3
+zi4
+zong1
+zong2
+zong3
+zong4
+zou1
+zou2
+zou3
+zou4
+zu1
+zu2
+zu3
+zuan1
+zuan3
+zuan4
+zui2
+zui3
+zui4
+zun1
+zuo
+zuo1
+zuo2
+zuo3
+zuo4
+{
+~
+¡
+¢
+£
+¥
+§
+¨
+©
+«
+®
+¯
+°
+±
+²
+³
+´
+µ
+·
+¹
+º
+»
+¼
+½
+¾
+¿
+À
+Á
+Â
+Ã
+Ä
+Å
+Æ
+Ç
+È
+É
+Ê
+Í
+Î
+Ñ
+Ó
+Ö
+×
+Ø
+Ú
+Ü
+Ý
+Þ
+ß
+à
+á
+â
+ã
+ä
+å
+æ
+ç
+è
+é
+ê
+ë
+ì
+í
+î
+ï
+ð
+ñ
+ò
+ó
+ô
+õ
+ö
+ø
+ù
+ú
+û
+ü
+ý
+Ā
+ā
+ă
+ą
+ć
+Č
+č
+Đ
+đ
+ē
+ė
+ę
+ě
+ĝ
+ğ
+ħ
+ī
+į
+İ
+ı
+Ł
+ł
+ń
+ņ
+ň
+ŋ
+Ō
+ō
+ő
+œ
+ř
+Ś
+ś
+Ş
+ş
+Š
+š
+Ť
+ť
+ũ
+ū
+ź
+Ż
+ż
+Ž
+ž
+ơ
+ư
+ǎ
+ǐ
+ǒ
+ǔ
+ǚ
+ș
+ț
+ɑ
+ɔ
+ɕ
+ə
+ɛ
+ɜ
+ɡ
+ɣ
+ɪ
+ɫ
+ɴ
+ɹ
+ɾ
+ʃ
+ʊ
+ʌ
+ʒ
+ʔ
+ʰ
+ʷ
+ʻ
+ʾ
+ʿ
+ˈ
+ː
+˙
+˜
+ˢ
+́
+̅
+Α
+Β
+Δ
+Ε
+Θ
+Κ
+Λ
+Μ
+Ξ
+Π
+Σ
+Τ
+Φ
+Χ
+Ψ
+Ω
+ά
+έ
+ή
+ί
+α
+β
+γ
+δ
+ε
+ζ
+η
+θ
+ι
+κ
+λ
+μ
+ν
+ξ
+ο
+π
+ρ
+ς
+σ
+τ
+υ
+φ
+χ
+ψ
+ω
+ϊ
+ό
+ύ
+ώ
+ϕ
+ϵ
+Ё
+А
+Б
+В
+Г
+Д
+Е
+Ж
+З
+И
+Й
+К
+Л
+М
+Н
+О
+П
+Р
+С
+Т
+У
+Ф
+Х
+Ц
+Ч
+Ш
+Щ
+Ы
+Ь
+Э
+Ю
+Я
+а
+б
+в
+г
+д
+е
+ж
+з
+и
+й
+к
+л
+м
+н
+о
+п
+р
+с
+т
+у
+ф
+х
+ц
+ч
+ш
+щ
+ъ
+ы
+ь
+э
+ю
+я
+ё
+і
+ְ
+ִ
+ֵ
+ֶ
+ַ
+ָ
+ֹ
+ּ
+־
+ׁ
+א
+ב
+ג
+ד
+ה
+ו
+ז
+ח
+ט
+י
+כ
+ל
+ם
+מ
+ן
+נ
+ס
+ע
+פ
+ק
+ר
+ש
+ת
+أ
+ب
+ة
+ت
+ج
+ح
+د
+ر
+ز
+س
+ص
+ط
+ع
+ق
+ك
+ل
+م
+ن
+ه
+و
+ي
+َ
+ُ
+ِ
+ْ
+ก
+ข
+ง
+จ
+ต
+ท
+น
+ป
+ย
+ร
+ว
+ส
+ห
+อ
+ฮ
+ั
+า
+ี
+ึ
+โ
+ใ
+ไ
+่
+้
+์
+ḍ
+Ḥ
+ḥ
+ṁ
+ṃ
+ṅ
+ṇ
+Ṛ
+ṛ
+Ṣ
+ṣ
+Ṭ
+ṭ
+ạ
+ả
+Ấ
+ấ
+ầ
+ậ
+ắ
+ằ
+ẻ
+ẽ
+ế
+ề
+ể
+ễ
+ệ
+ị
+ọ
+ỏ
+ố
+ồ
+ộ
+ớ
+ờ
+ở
+ụ
+ủ
+ứ
+ữ
+ἀ
+ἁ
+Ἀ
+ἐ
+ἔ
+ἰ
+ἱ
+ὀ
+ὁ
+ὐ
+ὲ
+ὸ
+ᾶ
+᾽
+ῆ
+ῇ
+ῶ
+‎
+‑
+‒
+–
+—
+―
+‖
+†
+‡
+•
+…
+‧
+‬
+′
+″
+⁄
+⁡
+⁰
+⁴
+⁵
+⁶
+⁷
+⁸
+⁹
+₁
+₂
+₃
+€
+₱
+₹
+₽
+℃
+ℏ
+ℓ
+№
+ℝ
+™
+⅓
+⅔
+⅛
+→
+∂
+∈
+∑
+−
+∗
+√
+∞
+∫
+≈
+≠
+≡
+≤
+≥
+⋅
+⋯
+█
+♪
+⟨
+⟩
+、
+。
+《
+》
+「
+」
+【
+】
+あ
+う
+え
+お
+か
+が
+き
+ぎ
+く
+ぐ
+け
+げ
+こ
+ご
+さ
+し
+じ
+す
+ず
+せ
+ぜ
+そ
+ぞ
+た
+だ
+ち
+っ
+つ
+で
+と
+ど
+な
+に
+ね
+の
+は
+ば
+ひ
+ぶ
+へ
+べ
+ま
+み
+む
+め
+も
+ゃ
+や
+ゆ
+ょ
+よ
+ら
+り
+る
+れ
+ろ
+わ
+を
+ん
+ァ
+ア
+ィ
+イ
+ウ
+ェ
+エ
+オ
+カ
+ガ
+キ
+ク
+ケ
+ゲ
+コ
+ゴ
+サ
+ザ
+シ
+ジ
+ス
+ズ
+セ
+ゾ
+タ
+ダ
+チ
+ッ
+ツ
+テ
+デ
+ト
+ド
+ナ
+ニ
+ネ
+ノ
+バ
+パ
+ビ
+ピ
+フ
+プ
+ヘ
+ベ
+ペ
+ホ
+ボ
+ポ
+マ
+ミ
+ム
+メ
+モ
+ャ
+ヤ
+ュ
+ユ
+ョ
+ヨ
+ラ
+リ
+ル
+レ
+ロ
+ワ
+ン
+・
+ー
+ㄋ
+ㄍ
+ㄎ
+ㄏ
+ㄓ
+ㄕ
+ㄚ
+ㄜ
+ㄟ
+ㄤ
+ㄥ
+ㄧ
+ㄱ
+ㄴ
+ㄷ
+ㄹ
+ㅁ
+ㅂ
+ㅅ
+ㅈ
+ㅍ
+ㅎ
+ㅏ
+ㅓ
+ㅗ
+ㅜ
+ㅡ
+ㅣ
+㗎
+가
+각
+간
+갈
+감
+갑
+갓
+갔
+강
+같
+개
+거
+건
+걸
+겁
+것
+겉
+게
+겠
+겨
+결
+겼
+경
+계
+고
+곤
+골
+곱
+공
+과
+관
+광
+교
+구
+국
+굴
+귀
+귄
+그
+근
+글
+금
+기
+긴
+길
+까
+깍
+깔
+깜
+깨
+께
+꼬
+꼭
+꽃
+꾸
+꿔
+끔
+끗
+끝
+끼
+나
+난
+날
+남
+납
+내
+냐
+냥
+너
+넘
+넣
+네
+녁
+년
+녕
+노
+녹
+놀
+누
+눈
+느
+는
+늘
+니
+님
+닙
+다
+닥
+단
+달
+닭
+당
+대
+더
+덕
+던
+덥
+데
+도
+독
+동
+돼
+됐
+되
+된
+될
+두
+둑
+둥
+드
+들
+등
+디
+따
+딱
+딸
+땅
+때
+떤
+떨
+떻
+또
+똑
+뚱
+뛰
+뜻
+띠
+라
+락
+란
+람
+랍
+랑
+래
+랜
+러
+런
+럼
+렇
+레
+려
+력
+렵
+렸
+로
+록
+롬
+루
+르
+른
+를
+름
+릉
+리
+릴
+림
+마
+막
+만
+많
+말
+맑
+맙
+맛
+매
+머
+먹
+멍
+메
+면
+명
+몇
+모
+목
+몸
+못
+무
+문
+물
+뭐
+뭘
+미
+민
+밌
+밑
+바
+박
+밖
+반
+받
+발
+밤
+밥
+방
+배
+백
+밸
+뱀
+버
+번
+벌
+벚
+베
+벼
+벽
+별
+병
+보
+복
+본
+볼
+봐
+봤
+부
+분
+불
+비
+빔
+빛
+빠
+빨
+뼈
+뽀
+뿅
+쁘
+사
+산
+살
+삼
+샀
+상
+새
+색
+생
+서
+선
+설
+섭
+섰
+성
+세
+셔
+션
+셨
+소
+속
+손
+송
+수
+숙
+순
+술
+숫
+숭
+숲
+쉬
+쉽
+스
+슨
+습
+슷
+시
+식
+신
+실
+싫
+심
+십
+싶
+싸
+써
+쓰
+쓴
+씌
+씨
+씩
+씬
+아
+악
+안
+않
+알
+야
+약
+얀
+양
+얘
+어
+언
+얼
+엄
+업
+없
+었
+엉
+에
+여
+역
+연
+염
+엽
+영
+옆
+예
+옛
+오
+온
+올
+옷
+옹
+와
+왔
+왜
+요
+욕
+용
+우
+운
+울
+웃
+워
+원
+월
+웠
+위
+윙
+유
+육
+윤
+으
+은
+을
+음
+응
+의
+이
+익
+인
+일
+읽
+임
+입
+있
+자
+작
+잔
+잖
+잘
+잡
+잤
+장
+재
+저
+전
+점
+정
+제
+져
+졌
+조
+족
+좀
+종
+좋
+죠
+주
+준
+줄
+중
+줘
+즈
+즐
+즘
+지
+진
+집
+짜
+짝
+쩌
+쪼
+쪽
+쫌
+쭈
+쯔
+찌
+찍
+차
+착
+찾
+책
+처
+천
+철
+체
+쳐
+쳤
+초
+촌
+추
+출
+춤
+춥
+춰
+치
+친
+칠
+침
+칩
+칼
+커
+켓
+코
+콩
+쿠
+퀴
+크
+큰
+큽
+키
+킨
+타
+태
+터
+턴
+털
+테
+토
+통
+투
+트
+특
+튼
+틀
+티
+팀
+파
+팔
+패
+페
+펜
+펭
+평
+포
+폭
+표
+품
+풍
+프
+플
+피
+필
+하
+학
+한
+할
+함
+합
+항
+해
+햇
+했
+행
+허
+험
+형
+혜
+호
+혼
+홀
+화
+회
+획
+후
+휴
+흐
+흔
+희
+히
+힘
+ﷺ
+ﷻ
+！
+，
+？
+�
+𠮶

2flow/models/downloads/F5TTS_v1_Base_no_zero_init/model_1250000.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:790d5b83e2afea3cc879fabfed58b2b4da214c882ef34513adfed82684a4c47f
+size 1348435761

2flow/patch/__init__.py ADDED Viewed

	@@ -0,0 +1,196 @@

+# SPDX-FileCopyrightText: Copyright (c) 2022-2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from .baichuan.model import BaichuanForCausalLM
+from .bert.model import (
+    BertForQuestionAnswering,
+    BertForSequenceClassification,
+    BertModel,
+    RobertaForQuestionAnswering,
+    RobertaForSequenceClassification,
+    RobertaModel,
+)
+from .bloom.model import BloomForCausalLM, BloomModel
+from .chatglm.config import ChatGLMConfig
+from .chatglm.model import ChatGLMForCausalLM, ChatGLMModel
+from .cogvlm.config import CogVLMConfig
+from .cogvlm.model import CogVLMForCausalLM
+from .commandr.model import CohereForCausalLM
+from .dbrx.config import DbrxConfig
+from .dbrx.model import DbrxForCausalLM
+from .deepseek_v1.model import DeepseekForCausalLM
+from .deepseek_v2.model import DeepseekV2ForCausalLM
+from .dit.model import DiT
+from .eagle.model import EagleForCausalLM
+from .enc_dec.model import DecoderModel, EncoderModel, WhisperEncoder
+from .f5tts.model import F5TTS
+from .falcon.config import FalconConfig
+from .falcon.model import FalconForCausalLM, FalconModel
+from .gemma.config import GEMMA2_ARCHITECTURE, GEMMA_ARCHITECTURE, GemmaConfig
+from .gemma.model import GemmaForCausalLM
+from .gpt.config import GPTConfig
+from .gpt.model import GPTForCausalLM, GPTModel
+from .gptj.config import GPTJConfig
+from .gptj.model import GPTJForCausalLM, GPTJModel
+from .gptneox.model import GPTNeoXForCausalLM, GPTNeoXModel
+from .grok.model import GrokForCausalLM
+from .llama.config import LLaMAConfig
+from .llama.model import LLaMAForCausalLM, LLaMAModel
+from .mamba.model import MambaForCausalLM
+from .medusa.config import MedusaConfig
+from .medusa.model import MedusaForCausalLm
+from .mllama.model import MLLaMAModel
+from .modeling_utils import PretrainedConfig, PretrainedModel, SpeculativeDecodingMode
+from .mpt.model import MPTForCausalLM, MPTModel
+from .nemotron_nas.model import DeciLMForCausalLM
+from .opt.model import OPTForCausalLM, OPTModel
+from .phi.model import PhiForCausalLM, PhiModel
+from .phi3.model import Phi3ForCausalLM, Phi3Model
+from .qwen.model import QWenForCausalLM
+from .recurrentgemma.model import RecurrentGemmaForCausalLM
+__all__ = [
+    "BertModel",
+    "BertForQuestionAnswering",
+    "BertForSequenceClassification",
+    "RobertaModel",
+    "RobertaForQuestionAnswering",
+    "RobertaForSequenceClassification",
+    "BloomModel",
+    "BloomForCausalLM",
+    "DiT",
+    "DeepseekForCausalLM",
+    "FalconConfig",
+    "DeepseekV2ForCausalLM",
+    "FalconForCausalLM",
+    "FalconModel",
+    "GPTConfig",
+    "GPTModel",
+    "GPTForCausalLM",
+    "OPTForCausalLM",
+    "OPTModel",
+    "LLaMAConfig",
+    "LLaMAForCausalLM",
+    "LLaMAModel",
+    "MedusaConfig",
+    "MedusaForCausalLm",
+    "GPTJConfig",
+    "GPTJModel",
+    "GPTJForCausalLM",
+    "GPTNeoXModel",
+    "GPTNeoXForCausalLM",
+    "PhiModel",
+    "PhiConfig",
+    "Phi3Model",
+    "Phi3Config",
+    "PhiForCausalLM",
+    "Phi3ForCausalLM",
+    "ChatGLMConfig",
+    "ChatGLMForCausalLM",
+    "ChatGLMModel",
+    "BaichuanForCausalLM",
+    "QWenConfigQWenForCausalLM",
+    "QWenModel",
+    "EncoderModel",
+    "DecoderModel",
+    "PretrainedConfig",
+    "PretrainedModel",
+    "WhisperEncoder",
+    "MambaForCausalLM",
+    "MambaConfig",
+    "MPTForCausalLM",
+    "MPTModel",
+    "SkyworkForCausalLM",
+    "GemmaConfig",
+    "GemmaForCausalLM",
+    "DbrxConfig",
+    "DbrxForCausalLM",
+    "RecurrentGemmaForCausalLM",
+    "CogVLMConfig",
+    "CogVLMForCausalLM",
+    "EagleForCausalLM",
+    "SpeculativeDecodingMode",
+    "CohereForCausalLM",
+    "MLLaMAModel",
+    "F5TTS",
+]
+MODEL_MAP = {
+    "GPT2LMHeadModel": GPTForCausalLM,
+    "GPT2LMHeadCustomModel": GPTForCausalLM,
+    "GPTBigCodeForCausalLM": GPTForCausalLM,
+    "Starcoder2ForCausalLM": GPTForCausalLM,
+    "FuyuForCausalLM": GPTForCausalLM,
+    "Kosmos2ForConditionalGeneration": GPTForCausalLM,
+    "JAISLMHeadModel": GPTForCausalLM,
+    "GPTForCausalLM": GPTForCausalLM,
+    "NemotronForCausalLM": GPTForCausalLM,
+    "OPTForCausalLM": OPTForCausalLM,
+    "BloomForCausalLM": BloomForCausalLM,
+    "RWForCausalLM": FalconForCausalLM,
+    "FalconForCausalLM": FalconForCausalLM,
+    "PhiForCausalLM": PhiForCausalLM,
+    "Phi3ForCausalLM": Phi3ForCausalLM,
+    "Phi3VForCausalLM": Phi3ForCausalLM,
+    "Phi3SmallForCausalLM": Phi3ForCausalLM,
+    "PhiMoEForCausalLM": Phi3ForCausalLM,
+    "MambaForCausalLM": MambaForCausalLM,
+    "GPTNeoXForCausalLM": GPTNeoXForCausalLM,
+    "GPTJForCausalLM": GPTJForCausalLM,
+    "MPTForCausalLM": MPTForCausalLM,
+    "GLMModel": ChatGLMForCausalLM,
+    "ChatGLMModel": ChatGLMForCausalLM,
+    "ChatGLMForCausalLM": ChatGLMForCausalLM,
+    "LlamaForCausalLM": LLaMAForCausalLM,
+    "ExaoneForCausalLM": LLaMAForCausalLM,
+    "MistralForCausalLM": LLaMAForCausalLM,
+    "MixtralForCausalLM": LLaMAForCausalLM,
+    "ArcticForCausalLM": LLaMAForCausalLM,
+    "Grok1ModelForCausalLM": GrokForCausalLM,
+    "InternLMForCausalLM": LLaMAForCausalLM,
+    "InternLM2ForCausalLM": LLaMAForCausalLM,
+    "MedusaForCausalLM": MedusaForCausalLm,
+    "BaichuanForCausalLM": BaichuanForCausalLM,
+    "BaiChuanForCausalLM": BaichuanForCausalLM,
+    "SkyworkForCausalLM": LLaMAForCausalLM,
+    GEMMA_ARCHITECTURE: GemmaForCausalLM,
+    GEMMA2_ARCHITECTURE: GemmaForCausalLM,
+    "QWenLMHeadModel": QWenForCausalLM,
+    "QWenForCausalLM": QWenForCausalLM,
+    "Qwen2ForCausalLM": QWenForCausalLM,
+    "Qwen2MoeForCausalLM": QWenForCausalLM,
+    "Qwen2ForSequenceClassification": QWenForCausalLM,
+    "Qwen2VLForConditionalGeneration": QWenForCausalLM,
+    "WhisperEncoder": WhisperEncoder,
+    "EncoderModel": EncoderModel,
+    "DecoderModel": DecoderModel,
+    "DbrxForCausalLM": DbrxForCausalLM,
+    "RecurrentGemmaForCausalLM": RecurrentGemmaForCausalLM,
+    "CogVLMForCausalLM": CogVLMForCausalLM,
+    "DiT": DiT,
+    "DeepseekForCausalLM": DeepseekForCausalLM,
+    "DeciLMForCausalLM": DeciLMForCausalLM,
+    "DeepseekV2ForCausalLM": DeepseekV2ForCausalLM,
+    "EagleForCausalLM": EagleForCausalLM,
+    "CohereForCausalLM": CohereForCausalLM,
+    "MllamaForConditionalGeneration": MLLaMAModel,
+    "BertForQuestionAnswering": BertForQuestionAnswering,
+    "BertForSequenceClassification": BertForSequenceClassification,
+    "BertModel": BertModel,
+    "RobertaModel": RobertaModel,
+    "RobertaForQuestionAnswering": RobertaForQuestionAnswering,
+    "RobertaForSequenceClassification": RobertaForSequenceClassification,
+    "F5TTS": F5TTS,
+}

2flow/patch/f5tts/model.py ADDED Viewed

	@@ -0,0 +1,222 @@

+from __future__ import annotations
+import os
+import sys
+from collections import OrderedDict
+import tensorrt as trt
+from tensorrt_llm._common import default_net
+from ..._utils import str_dtype_to_trt
+from ...functional import Tensor, concat
+from ...layers import Linear
+from ...module import Module, ModuleList
+from ...plugin import current_all_reduce_helper
+from ..modeling_utils import PretrainedConfig, PretrainedModel
+from .modules import AdaLayerNormZero_Final, ConvPositionEmbedding, DiTBlock, TimestepEmbedding
+current_file_path = os.path.abspath(__file__)
+parent_dir = os.path.dirname(current_file_path)
+sys.path.append(parent_dir)
+class InputEmbedding(Module):
+    def __init__(self, mel_dim, text_dim, out_dim):
+        super().__init__()
+        self.proj = Linear(mel_dim * 2 + text_dim, out_dim)
+        self.conv_pos_embed = ConvPositionEmbedding(dim=out_dim)
+    def forward(self, x, cond):
+        x = self.proj(concat([x, cond], dim=-1))
+        return self.conv_pos_embed(x) + x
+class F5TTS(PretrainedModel):
+    def __init__(self, config: PretrainedConfig):
+        super().__init__(config)
+        self.dtype = str_dtype_to_trt(config.dtype)
+        self.time_embed = TimestepEmbedding(config.hidden_size)
+        self.input_embed = InputEmbedding(config.mel_dim, config.text_dim, config.hidden_size)
+        self.dim = config.hidden_size
+        self.depth = config.num_hidden_layers
+        self.transformer_blocks = ModuleList(
+            [
+                DiTBlock(
+                    dim=self.dim,
+                    heads=config.num_attention_heads,
+                    dim_head=config.dim_head,
+                    ff_mult=config.ff_mult,
+                    dropout=config.dropout,
+                )
+                for _ in range(self.depth)
+            ]
+        )
+        self.norm_out = AdaLayerNormZero_Final(config.hidden_size)  # final modulation
+        self.proj_out = Linear(config.hidden_size, config.mel_dim)
+    def forward(
+        self,
+        noise,  # nosied input audio
+        cond,  # masked cond audio
+        time,  # time step
+        rope_cos,
+        rope_sin,
+        input_lengths,
+        scale=1.0,
+    ):
+        t = self.time_embed(time)
+        x = self.input_embed(noise, cond)
+        for block in self.transformer_blocks:
+            x = block(x, t, rope_cos=rope_cos, rope_sin=rope_sin, input_lengths=input_lengths, scale=scale)
+        denoise = self.proj_out(self.norm_out(x, t))
+        denoise.mark_output("denoised", self.dtype)
+        return denoise
+    def prepare_inputs(self, **kwargs):
+        max_batch_size = kwargs["max_batch_size"]
+        batch_size_range = [2, 2, max_batch_size]
+        mel_size = 100
+        max_seq_len = 3000
+        num_frames_range = [200, 2 * max_seq_len, max_seq_len * max_batch_size]
+        hidden_size = 512
+        concat_feature_dim = mel_size + hidden_size
+        freq_embed_dim = 256
+        head_dim = 64
+        mapping = self.config.mapping
+        if mapping.tp_size > 1:
+            current_all_reduce_helper().set_workspace_tensor(mapping, 1)
+        if default_net().plugin_config.remove_input_padding:
+            noise = Tensor(
+                name="noise",
+                dtype=self.dtype,
+                shape=[-1, mel_size],
+                dim_range=OrderedDict(
+                    [
+                        ("num_frames", [num_frames_range]),
+                        ("n_mels", [mel_size]),
+                    ]
+                ),
+            )
+            cond = Tensor(
+                name="cond",
+                dtype=self.dtype,
+                shape=[-1, concat_feature_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("num_frames", [num_frames_range]),
+                        ("embeded_length", [concat_feature_dim]),
+                    ]
+                ),
+            )
+            time = Tensor(
+                name="time",
+                dtype=self.dtype,
+                shape=[-1, freq_embed_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("num_frames", [num_frames_range]),
+                        ("freq_dim", [freq_embed_dim]),
+                    ]
+                ),
+            )
+            rope_cos = Tensor(
+                name="rope_cos",
+                dtype=self.dtype,
+                shape=[-1, head_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("num_frames", [num_frames_range]),
+                        ("head_dim", [head_dim]),
+                    ]
+                ),
+            )
+            rope_sin = Tensor(
+                name="rope_sin",
+                dtype=self.dtype,
+                shape=[-1, head_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("num_frames", [num_frames_range]),
+                        ("head_dim", [head_dim]),
+                    ]
+                ),
+            )
+        else:
+            noise = Tensor(
+                name="noise",
+                dtype=self.dtype,
+                shape=[-1, -1, mel_size],
+                dim_range=OrderedDict(
+                    [
+                        ("batch_size", [batch_size_range]),
+                        ("max_duratuion", [[100, max_seq_len // 2, max_seq_len]]),
+                        ("n_mels", [mel_size]),
+                    ]
+                ),
+            )
+            cond = Tensor(
+                name="cond",
+                dtype=self.dtype,
+                shape=[-1, -1, concat_feature_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("batch_size", [batch_size_range]),
+                        ("max_duratuion", [[100, max_seq_len // 2, max_seq_len]]),
+                        ("embeded_length", [concat_feature_dim]),
+                    ]
+                ),
+            )
+            time = Tensor(
+                name="time",
+                dtype=self.dtype,
+                shape=[-1, freq_embed_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("batch_size", [batch_size_range]),
+                        ("freq_dim", [freq_embed_dim]),
+                    ]
+                ),
+            )
+            rope_cos = Tensor(
+                name="rope_cos",
+                dtype=self.dtype,
+                shape=[-1, -1, head_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("batch_size", [batch_size_range]),
+                        ("max_duratuion", [[100, max_seq_len // 2, max_seq_len]]),
+                        ("head_dim", [head_dim]),
+                    ]
+                ),
+            )
+            rope_sin = Tensor(
+                name="rope_sin",
+                dtype=self.dtype,
+                shape=[-1, -1, head_dim],
+                dim_range=OrderedDict(
+                    [
+                        ("batch_size", [batch_size_range]),
+                        ("max_duratuion", [[100, max_seq_len // 2, max_seq_len]]),
+                        ("head_dim", [head_dim]),
+                    ]
+                ),
+            )
+        input_lengths = Tensor(
+            name="input_lengths",
+            dtype=trt.int32,
+            shape=[-1],
+            dim_range=OrderedDict([("batch_size", [batch_size_range])]),
+        )
+        return {
+            "noise": noise,
+            "cond": cond,
+            "time": time,
+            "rope_cos": rope_cos,
+            "rope_sin": rope_sin,
+            "input_lengths": input_lengths,
+        }

2flow/patch/f5tts/modules.py ADDED Viewed

	@@ -0,0 +1,447 @@

+from __future__ import annotations
+import math
+from typing import Optional
+import numpy as np
+import torch
+import torch.nn.functional as F
+from tensorrt_llm._common import default_net
+from ..._utils import str_dtype_to_trt, trt_dtype_to_np
+from ...functional import (
+    Tensor,
+    bert_attention,
+    cast,
+    chunk,
+    concat,
+    constant,
+    expand,
+    expand_dims,
+    expand_dims_like,
+    expand_mask,
+    gelu,
+    matmul,
+    permute,
+    shape,
+    silu,
+    slice,
+    softmax,
+    squeeze,
+    unsqueeze,
+    view,
+)
+from ...layers import ColumnLinear, Conv1d, LayerNorm, Linear, Mish, RowLinear
+from ...module import Module
+class FeedForward(Module):
+    def __init__(self, dim, dim_out=None, mult=4, dropout=0.0):
+        super().__init__()
+        inner_dim = int(dim * mult)
+        dim_out = dim_out if dim_out is not None else dim
+        self.project_in = Linear(dim, inner_dim)
+        self.ff = Linear(inner_dim, dim_out)
+    def forward(self, x):
+        return self.ff(gelu(self.project_in(x)))
+class AdaLayerNormZero(Module):
+    def __init__(self, dim):
+        super().__init__()
+        self.linear = Linear(dim, dim * 6)
+        self.norm = LayerNorm(dim, elementwise_affine=False, eps=1e-6)
+    def forward(self, x, emb=None):
+        emb = self.linear(silu(emb))
+        shift_msa, scale_msa, gate_msa, shift_mlp, scale_mlp, gate_mlp = chunk(emb, 6, dim=1)
+        x = self.norm(x)
+        ones = constant(np.ones(1, dtype=np.float32)).cast(x.dtype)
+        if default_net().plugin_config.remove_input_padding:
+            x = x * (ones + scale_msa) + shift_msa
+        else:
+            x = x * (ones + unsqueeze(scale_msa, 1)) + unsqueeze(shift_msa, 1)
+        return x, gate_msa, shift_mlp, scale_mlp, gate_mlp
+class AdaLayerNormZero_Final(Module):
+    def __init__(self, dim):
+        super().__init__()
+        self.linear = Linear(dim, dim * 2)
+        self.norm = LayerNorm(dim, elementwise_affine=False, eps=1e-6)
+    def forward(self, x, emb):
+        emb = self.linear(silu(emb))
+        scale, shift = chunk(emb, 2, dim=1)
+        ones = constant(np.ones(1, dtype=np.float32)).cast(x.dtype)
+        if default_net().plugin_config.remove_input_padding:
+            x = self.norm(x) * (ones + scale) + shift
+        else:
+            x = self.norm(x) * unsqueeze((ones + scale), 1)
+            x = x + unsqueeze(shift, 1)
+        return x
+class ConvPositionEmbedding(Module):
+    def __init__(self, dim, kernel_size=31, groups=16):
+        super().__init__()
+        assert kernel_size % 2 != 0
+        self.conv1d1 = Conv1d(dim, dim, kernel_size, groups=groups, padding=kernel_size // 2)
+        self.conv1d2 = Conv1d(dim, dim, kernel_size, groups=groups, padding=kernel_size // 2)
+        self.mish = Mish()
+    def forward(self, x, mask=None):  # noqa: F722
+        if default_net().plugin_config.remove_input_padding:
+            x = unsqueeze(x, 0)
+        x = permute(x, [0, 2, 1])
+        x = self.mish(self.conv1d2(self.mish(self.conv1d1(x))))
+        out = permute(x, [0, 2, 1])
+        if default_net().plugin_config.remove_input_padding:
+            out = squeeze(out, 0)
+        return out
+class Attention(Module):
+    def __init__(
+        self,
+        processor: AttnProcessor,
+        dim: int,
+        heads: int = 16,
+        dim_head: int = 64,
+        dropout: float = 0.0,
+        context_dim: Optional[int] = None,  # if not None -> joint attention
+        context_pre_only=None,
+    ):
+        super().__init__()
+        if not hasattr(F, "scaled_dot_product_attention"):
+            raise ImportError("Attention equires PyTorch 2.0, to use it, please upgrade PyTorch to 2.0.")
+        self.processor = processor
+        self.dim = dim  # hidden_size
+        self.heads = heads
+        self.inner_dim = dim_head * heads
+        self.dropout = dropout
+        self.attention_head_size = dim_head
+        self.context_dim = context_dim
+        self.context_pre_only = context_pre_only
+        self.tp_size = 1
+        self.num_attention_heads = heads // self.tp_size
+        self.num_attention_kv_heads = heads // self.tp_size  # 8
+        self.dtype = str_dtype_to_trt("float32")
+        self.attention_hidden_size = self.attention_head_size * self.num_attention_heads
+        self.to_q = ColumnLinear(
+            dim,
+            self.tp_size * self.num_attention_heads * self.attention_head_size,
+            bias=True,
+            dtype=self.dtype,
+            tp_group=None,
+            tp_size=self.tp_size,
+        )
+        self.to_k = ColumnLinear(
+            dim,
+            self.tp_size * self.num_attention_heads * self.attention_head_size,
+            bias=True,
+            dtype=self.dtype,
+            tp_group=None,
+            tp_size=self.tp_size,
+        )
+        self.to_v = ColumnLinear(
+            dim,
+            self.tp_size * self.num_attention_heads * self.attention_head_size,
+            bias=True,
+            dtype=self.dtype,
+            tp_group=None,
+            tp_size=self.tp_size,
+        )
+        if self.context_dim is not None:
+            self.to_k_c = Linear(context_dim, self.inner_dim)
+            self.to_v_c = Linear(context_dim, self.inner_dim)
+            if self.context_pre_only is not None:
+                self.to_q_c = Linear(context_dim, self.inner_dim)
+        self.to_out = RowLinear(
+            self.tp_size * self.num_attention_heads * self.attention_head_size,
+            dim,
+            bias=True,
+            dtype=self.dtype,
+            tp_group=None,
+            tp_size=self.tp_size,
+        )
+        if self.context_pre_only is not None and not self.context_pre_only:
+            self.to_out_c = Linear(self.inner_dim, dim)
+    def forward(
+        self,
+        x,  # noised input x
+        rope_cos,
+        rope_sin,
+        input_lengths,
+        c=None,  # context c
+        scale=1.0,
+        rope=None,
+        c_rope=None,  # rotary position embedding for c
+    ) -> torch.Tensor:
+        if c is not None:
+            return self.processor(self, x, c=c, input_lengths=input_lengths, scale=scale, rope=rope, c_rope=c_rope)
+        else:
+            return self.processor(
+                self, x, rope_cos=rope_cos, rope_sin=rope_sin, input_lengths=input_lengths, scale=scale
+            )
+def rotate_every_two_3dim(tensor: Tensor) -> Tensor:
+    shape_tensor = concat(
+        [shape(tensor, i) / 2 if i == (tensor.ndim() - 1) else shape(tensor, i) for i in range(tensor.ndim())]
+    )
+    if default_net().plugin_config.remove_input_padding:
+        assert tensor.ndim() == 2
+        x1 = slice(tensor, [0, 0], shape_tensor, [1, 2])
+        x2 = slice(tensor, [0, 1], shape_tensor, [1, 2])
+        x1 = expand_dims(x1, 2)
+        x2 = expand_dims(x2, 2)
+        zero = constant(np.ascontiguousarray(np.zeros([1], dtype=trt_dtype_to_np(tensor.dtype))))
+        x2 = zero - x2
+        x = concat([x2, x1], 2)
+        out = view(x, concat([shape(x, 0), shape(x, 1) * 2]))
+    else:
+        assert tensor.ndim() == 3
+        x1 = slice(tensor, [0, 0, 0], shape_tensor, [1, 1, 2])
+        x2 = slice(tensor, [0, 0, 1], shape_tensor, [1, 1, 2])
+        x1 = expand_dims(x1, 3)
+        x2 = expand_dims(x2, 3)
+        zero = constant(np.ascontiguousarray(np.zeros([1], dtype=trt_dtype_to_np(tensor.dtype))))
+        x2 = zero - x2
+        x = concat([x2, x1], 3)
+        out = view(x, concat([shape(x, 0), shape(x, 1), shape(x, 2) * 2]))
+    return out
+# def apply_rotary_pos_emb_3dim(x, rope_cos, rope_sin):
+#     if default_net().plugin_config.remove_input_padding:
+#         rot_dim = shape(rope_cos, -1)  # 64
+#         new_t_shape = concat([shape(x, 0), rot_dim])  # (-1, 64)
+#         x_ = slice(x, [0, 0], new_t_shape, [1, 1])
+#         end_dim = shape(x, -1) - shape(rope_cos, -1)
+#         new_t_unrotated_shape = concat([shape(x, 0), end_dim])  # (2, -1, 960)
+#         x_unrotated = slice(x, concat([0, rot_dim]), new_t_unrotated_shape, [1, 1])
+#         out = concat([x_ * rope_cos + rotate_every_two_3dim(x_) * rope_sin, x_unrotated], dim=-1)
+#     else:
+#         rot_dim = shape(rope_cos, 2)  # 64
+#         new_t_shape = concat([shape(x, 0), shape(x, 1), rot_dim])  # (2, -1, 64)
+#         x_ = slice(x, [0, 0, 0], new_t_shape, [1, 1, 1])
+#         end_dim = shape(x, 2) - shape(rope_cos, 2)
+#         new_t_unrotated_shape = concat([shape(x, 0), shape(x, 1), end_dim])  # (2, -1, 960)
+#         x_unrotated = slice(x, concat([0, 0, rot_dim]), new_t_unrotated_shape, [1, 1, 1])
+#         out = concat([x_ * rope_cos + rotate_every_two_3dim(x_) * rope_sin, x_unrotated], dim=-1)
+#     return out
+def apply_rotary_pos_emb_3dim(x, rope_cos, rope_sin):
+    """
+    Apply RoPE for each block (like 64 dims) across all heads.
+    Supports both normal and remove_input_padding=True mode.
+    """
+    if default_net().plugin_config.remove_input_padding:
+        # For [N, D] input
+        full_dim = shape(x, 1)
+        block_size = shape(rope_cos, 1)
+        out_blocks = []
+        for i in range(16):
+            start = i * 64
+            curr_shape = concat([shape(x, 0), block_size])
+            x_block = slice(x, [0, start], curr_shape, [1, 1])
+            cos_block = slice(rope_cos, [0, start], curr_shape, [1, 1])
+            sin_block = slice(rope_sin, [0, start], curr_shape, [1, 1])
+            rotated = rotate_every_two_3dim(x_block)
+            block_out = x_block * cos_block + rotated * sin_block
+            out_blocks.append(block_out)
+        out = concat(out_blocks, dim=-1)
+    else:
+        # For [B, N, D] input
+        pieces = []
+        rot_dim = shape(rope_cos, 2)
+        full_dim = shape(x, 2)
+        new_t_shape = concat([shape(x, 0), shape(x, 1), rot_dim])
+        for i in range(16):
+            x_slice = slice(x, [0, 0, i*64], new_t_shape, [1, 1, 1])
+            rotated_slice = x_slice * rope_cos + rotate_every_two_3dim(x_slice) * rope_sin
+            pieces.append(rotated_slice)
+        out = concat(pieces, dim=-1)
+    return out
+class AttnProcessor:
+    def __init__(self):
+        pass
+    def __call__(
+        self,
+        attn,
+        x,  # noised input x
+        rope_cos,
+        rope_sin,
+        input_lengths,
+        scale=1.0,
+        rope=None,
+    ) -> torch.FloatTensor:
+        query = attn.to_q(x)
+        key = attn.to_k(x)
+        value = attn.to_v(x)
+        # k,v,q all (2,1226,1024)
+        query = apply_rotary_pos_emb_3dim(query, rope_cos, rope_sin)
+        key = apply_rotary_pos_emb_3dim(key, rope_cos, rope_sin)
+        # attention
+        inner_dim = key.shape[-1]
+        norm_factor = math.sqrt(attn.attention_head_size)
+        q_scaling = 1.0 / norm_factor
+        mask = None
+        if not default_net().plugin_config.remove_input_padding:
+            N = shape(x, 1)
+            B = shape(x, 0)
+            seq_len_2d = concat([1, N])
+            max_position_embeddings = 4096
+            # create position ids
+            position_ids_buffer = constant(np.expand_dims(np.arange(max_position_embeddings).astype(np.int32), 0))
+            tmp_position_ids = slice(position_ids_buffer, starts=[0, 0], sizes=seq_len_2d)
+            tmp_position_ids = expand(tmp_position_ids, concat([B, N]))  # BxL
+            tmp_input_lengths = unsqueeze(input_lengths, 1)  # Bx1
+            tmp_input_lengths = expand(tmp_input_lengths, concat([B, N]))  # BxL
+            mask = tmp_position_ids < tmp_input_lengths  # BxL
+            mask = mask.cast("int32")
+        if default_net().plugin_config.bert_attention_plugin:
+            qkv = concat([query, key, value], dim=-1)
+            # TRT plugin mode
+            assert input_lengths is not None
+            if default_net().plugin_config.remove_input_padding:
+                qkv = qkv.view(concat([-1, 3 * inner_dim]))
+                max_input_length = constant(
+                    np.zeros(
+                        [
+                            2048,
+                        ],
+                        dtype=np.int32,
+                    )
+                )
+            else:
+                max_input_length = None
+            context = bert_attention(
+                qkv,
+                input_lengths,
+                attn.num_attention_heads,
+                attn.attention_head_size,
+                q_scaling=q_scaling,
+                max_input_length=max_input_length,
+            )
+        else:
+            assert not default_net().plugin_config.remove_input_padding
+            def transpose_for_scores(x):
+                new_x_shape = concat([shape(x, 0), shape(x, 1), attn.num_attention_heads, attn.attention_head_size])
+                y = x.view(new_x_shape)
+                y = y.transpose(1, 2)
+                return y
+            def transpose_for_scores_k(x):
+                new_x_shape = concat([shape(x, 0), shape(x, 1), attn.num_attention_heads, attn.attention_head_size])
+                y = x.view(new_x_shape)
+                y = y.permute([0, 2, 3, 1])
+                return y
+            query = transpose_for_scores(query)
+            key = transpose_for_scores_k(key)
+            value = transpose_for_scores(value)
+            attention_scores = matmul(query, key, use_fp32_acc=False)
+            if mask is not None:
+                attention_mask = expand_mask(mask, shape(query, 2))
+                attention_mask = cast(attention_mask, attention_scores.dtype)
+                attention_scores = attention_scores + attention_mask
+            attention_probs = softmax(attention_scores, dim=-1)
+            context = matmul(attention_probs, value, use_fp32_acc=False).transpose(1, 2)
+            context = context.view(concat([shape(context, 0), shape(context, 1), attn.attention_hidden_size]))
+        context = attn.to_out(context)
+        if mask is not None:
+            mask = mask.view(concat([shape(mask, 0), shape(mask, 1), 1]))
+            mask = expand_dims_like(mask, context)
+            mask = cast(mask, context.dtype)
+            context = context * mask
+        return context
+# DiT Block
+class DiTBlock(Module):
+    def __init__(self, dim, heads, dim_head, ff_mult=2, dropout=0.1):
+        super().__init__()
+        self.attn_norm = AdaLayerNormZero(dim)
+        self.attn = Attention(
+            processor=AttnProcessor(),
+            dim=dim,
+            heads=heads,
+            dim_head=dim_head,
+            dropout=dropout,
+        )
+        self.ff_norm = LayerNorm(dim, elementwise_affine=False, eps=1e-6)
+        self.ff = FeedForward(dim=dim, mult=ff_mult, dropout=dropout)
+    def forward(
+        self, x, t, rope_cos, rope_sin, input_lengths, scale=1.0, rope=ModuleNotFoundError
+    ):  # x: noised input, t: time embedding
+        # pre-norm & modulation for attention input
+        norm, gate_msa, shift_mlp, scale_mlp, gate_mlp = self.attn_norm(x, emb=t)
+        # attention
+        # norm ----> (2,1226,1024)
+        attn_output = self.attn(x=norm, rope_cos=rope_cos, rope_sin=rope_sin, input_lengths=input_lengths, scale=scale)
+        # process attention output for input x
+        if default_net().plugin_config.remove_input_padding:
+            x = x + gate_msa * attn_output
+        else:
+            x = x + unsqueeze(gate_msa, 1) * attn_output
+        ones = constant(np.ones(1, dtype=np.float32)).cast(x.dtype)
+        if default_net().plugin_config.remove_input_padding:
+            norm = self.ff_norm(x) * (ones + scale_mlp) + shift_mlp
+        else:
+            norm = self.ff_norm(x) * (ones + unsqueeze(scale_mlp, 1)) + unsqueeze(shift_mlp, 1)
+            # norm = self.ff_norm(x) * (ones + scale_mlp) + shift_mlp
+        ff_output = self.ff(norm)
+        if default_net().plugin_config.remove_input_padding:
+            x = x + gate_mlp * ff_output
+        else:
+            x = x + unsqueeze(gate_mlp, 1) * ff_output
+        return x
+class TimestepEmbedding(Module):
+    def __init__(self, dim, freq_embed_dim=256, dtype=None):
+        super().__init__()
+        # self.time_embed = SinusPositionEmbedding(freq_embed_dim)
+        self.mlp1 = Linear(freq_embed_dim, dim, bias=True, dtype=dtype)
+        self.mlp2 = Linear(dim, dim, bias=True, dtype=dtype)
+    def forward(self, timestep):
+        t_freq = self.mlp1(timestep)
+        t_freq = silu(t_freq)
+        t_emb = self.mlp2(t_freq)
+        return t_emb

2flow/requirements.txt ADDED Viewed

	@@ -0,0 +1,5 @@

+conv-stft
+vocos
+safetensors
+tensorrt_llm
+onnxscript

2flow/scripts/build.sh ADDED Viewed

	@@ -0,0 +1,2 @@


1	+ docker build -t tired:lastest .
2	+

2flow/scripts/f5/build_engine.sh ADDED Viewed

	@@ -0,0 +1,5 @@

+trtllm-build \
+    --checkpoint_dir ./models/pre_engine/tts \
+    --max_batch_size 8 \
+    --output_dir ./models/engine/tts \
+    --remove_input_padding "disable"

2flow/scripts/f5/fix_lib.py ADDED Viewed

	@@ -0,0 +1,32 @@

+import shutil
+import tensorrt_llm
+from pathlib import Path
+trtllm_path = Path(tensorrt_llm.__file__).parent
+target_dir = trtllm_path / "models"
+print(f"TensorRT-LLM path: {trtllm_path}")
+print(f"Target models directory: {target_dir}")
+target_dir.mkdir(parents=True, exist_ok=True)
+patch_dir = Path("./patch")
+patch_files = list(patch_dir.glob('*'))
+if patch_files:
+    print(f"Copying {len(patch_files)} patch file(s) to tensorrt_llm/models")
+    for patch_file in patch_files:
+        target_path = target_dir / patch_file.name
+        if patch_file.is_file():
+            shutil.copy2(patch_file, target_path)
+            print(f"  Copied: {patch_file.name}")
+        elif patch_file.is_dir():
+            if target_path.exists():
+                shutil.rmtree(target_path)
+            shutil.copytree(patch_file, target_path)
+            print(f"  Copied directory: {patch_file.name}")
+    print(f"✓ Patch files copied successfully")
+else:
+    print(f"⚠ No patch files found in {patch_dir}")

2flow/scripts/f5/pre_build_engine.sh ADDED Viewed

	@@ -0,0 +1,4 @@

+python3 -m utils.tts.convert_checkpoint \
+    --timm_ckpt ./models/downloads/F5TTS_v1_Base/model_1250000.safetensors \
+    --output_dir ./models/pre_engine/tts \
+    --model_name F5TTS_v1_Base

2flow/scripts/init.sh ADDED Viewed

	@@ -0,0 +1,6 @@

+docker run -it --rm \
+  --gpus all \
+  -v /mnt/hoang.dinh/code/2flow:/workspace/2flow \
+  -w /workspace/2flow \
+  tired:lastest \
+  bash

2flow/scripts/vocoder/build_engine.sh ADDED Viewed

	@@ -0,0 +1,3 @@

+bash scripts/vocoder/export_vocos_trt.sh \
+    ./models/pre_engine/tts/vocos_vocoder.onnx \
+    ./models/engine/tts/vocos_vocoder.plan

2flow/scripts/vocoder/export_vocos_trt.sh ADDED Viewed

	@@ -0,0 +1,43 @@

+#!/bin/bash
+# Copyright (c) 2025, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+TRTEXEC="/usr/src/tensorrt/bin/trtexec"
+ONNX_PATH=$1
+ENGINE_PATH=$2
+echo "ONNX_PATH: $ONNX_PATH"
+echo "ENGINE_PATH: $ENGINE_PATH"
+PRECISION="fp32"
+MIN_BATCH_SIZE=1
+OPT_BATCH_SIZE=1
+MAX_BATCH_SIZE=8
+MIN_INPUT_LENGTH=1
+OPT_INPUT_LENGTH=1000
+MAX_INPUT_LENGTH=3000
+MEL_MIN_SHAPE="${MIN_BATCH_SIZE}x100x${MIN_INPUT_LENGTH}"
+MEL_OPT_SHAPE="${OPT_BATCH_SIZE}x100x${OPT_INPUT_LENGTH}"
+MEL_MAX_SHAPE="${MAX_BATCH_SIZE}x100x${MAX_INPUT_LENGTH}"
+${TRTEXEC} \
+    --minShapes="mel:${MEL_MIN_SHAPE}" \
+    --optShapes="mel:${MEL_OPT_SHAPE}" \
+    --maxShapes="mel:${MEL_MAX_SHAPE}" \
+    --onnx=${ONNX_PATH} \
+    --saveEngine=${ENGINE_PATH}

2flow/scripts/vocoder/pre_build_engine.sh ADDED Viewed

	@@ -0,0 +1,3 @@

+python3 -m utils.tts.export_vocoder_to_onnx \
+    --vocoder vocos \
+    --output-path ./models/pre_engine/tts/vocos_vocoder.onnx

2flow/services/triton/f5_tts_triton_server/f5_tts/1/f5_tts_trtllm.py ADDED Viewed

	@@ -0,0 +1,486 @@

+import math
+import os
+import time
+from functools import wraps
+from typing import List, Optional
+import safetensors.torch
+import tensorrt as trt
+import tensorrt_llm
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from tensorrt_llm._utils import str_dtype_to_torch, trt_dtype_to_torch
+from tensorrt_llm.logger import logger
+from tensorrt_llm.runtime.session import Session
+def remove_tensor_padding(input_tensor, input_tensor_lengths=None):
+    # Audio tensor case: batch, seq_len, feature_len
+    # position_ids case: batch, seq_len
+    assert input_tensor_lengths is not None, "input_tensor_lengths must be provided for 3D input_tensor"
+    # Initialize a list to collect valid sequences
+    valid_sequences = []
+    for i in range(input_tensor.shape[0]):
+        valid_length = input_tensor_lengths[i]
+        valid_sequences.append(input_tensor[i, :valid_length])
+    # Concatenate all valid sequences along the batch dimension
+    output_tensor = torch.cat(valid_sequences, dim=0).contiguous()
+    return output_tensor
+# class TextEmbedding(nn.Module):
+#     def __init__(self, text_num_embeds, text_dim, conv_layers=0, conv_mult=2, precompute_max_pos=4096):
+#         super().__init__()
+#         self.text_embed = nn.Embedding(text_num_embeds + 1, text_dim)  # use 0 as filler token
+#         self.register_buffer("freqs_cis", precompute_freqs_cis(text_dim, precompute_max_pos), persistent=False)
+#         self.text_blocks = nn.Sequential(*[ConvNeXtV2Block(text_dim, text_dim * conv_mult) for _ in range(conv_layers)])
+#     def forward(self, text):
+#         # only keep tensors with value not -1
+#         text_mask = text != -1
+#         text_pad_cut_off_index = text_mask.sum(dim=1).max()
+#         text = text[:, :text_pad_cut_off_index]
+#         text = self.text_embed(text)
+#         text = text + self.freqs_cis[: text.shape[1], :]
+#         for block in self.text_blocks:
+#             text = block(text)
+#         # padding text to the original length
+#         # text shape: B,seq_len,C
+#         # pad at the second dimension
+#         text = F.pad(text, (0, 0, 0, text_mask.shape[1] - text.shape[1], 0, 0), value=0)
+#         return text
+class TextEmbedding(nn.Module):
+    def __init__(self, text_num_embeds, text_dim, conv_layers=0, conv_mult=2, precompute_max_pos=4096):
+        super().__init__()
+        self.text_embed = nn.Embedding(text_num_embeds + 1, text_dim)  # use 0 as filler token
+        self.register_buffer("freqs_cis", precompute_freqs_cis(text_dim, precompute_max_pos), persistent=False)
+        self.text_blocks = nn.Sequential(*[ConvNeXtV2Block(text_dim, text_dim * conv_mult) for _ in range(conv_layers)])
+    def forward(self, text):
+        # only keep tensors with value not -1
+        text_mask = text != -1
+        text_pad_cut_off_index = text_mask.sum(dim=1).max()
+        text_mask_cutoff = text  == 0
+        text = text[:, :text_pad_cut_off_index]
+        text = self.text_embed(text)
+        text = text + self.freqs_cis[:text.shape[1], :]
+        text = text.masked_fill(text_mask_cutoff.unsqueeze(-1).expand(-1, -1, text.size(-1)), 0.0)
+        for block in self.text_blocks:
+            text = block(text)
+            text = text.masked_fill(text_mask_cutoff.unsqueeze(-1).expand(-1, -1, text.size(-1)), 0.0)
+        # padding text back to original length
+        text = F.pad(text, (0, 0, 0, text_mask.shape[1] - text.shape[1], 0, 0), value=0)
+        return text
+class GRN(nn.Module):
+    def __init__(self, dim):
+        super().__init__()
+        self.gamma = nn.Parameter(torch.zeros(1, 1, dim))
+        self.beta = nn.Parameter(torch.zeros(1, 1, dim))
+    def forward(self, x):
+        Gx = torch.norm(x, p=2, dim=1, keepdim=True)
+        Nx = Gx / (Gx.mean(dim=-1, keepdim=True) + 1e-6)
+        return self.gamma * (x * Nx) + self.beta + x
+class ConvNeXtV2Block(nn.Module):
+    def __init__(
+        self,
+        dim: int,
+        intermediate_dim: int,
+        dilation: int = 1,
+    ):
+        super().__init__()
+        padding = (dilation * (7 - 1)) // 2
+        self.dwconv = nn.Conv1d(
+            dim, dim, kernel_size=7, padding=padding, groups=dim, dilation=dilation
+        )  # depthwise conv
+        self.norm = nn.LayerNorm(dim, eps=1e-6)
+        self.pwconv1 = nn.Linear(dim, intermediate_dim)  # pointwise/1x1 convs, implemented with linear layers
+        self.act = nn.GELU()
+        self.grn = GRN(intermediate_dim)
+        self.pwconv2 = nn.Linear(intermediate_dim, dim)
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        residual = x
+        x = x.transpose(1, 2)  # b n d -> b d n
+        x = self.dwconv(x)
+        x = x.transpose(1, 2)  # b d n -> b n d
+        x = self.norm(x)
+        x = self.pwconv1(x)
+        x = self.act(x)
+        x = self.grn(x)
+        x = self.pwconv2(x)
+        return residual + x
+def precompute_freqs_cis(dim: int, end: int, theta: float = 10000.0, theta_rescale_factor=1.0):
+    # proposed by reddit user bloc97, to rescale rotary embeddings to longer sequence length without fine-tuning
+    # has some connection to NTK literature
+    # https://www.reddit.com/r/LocalLLaMA/comments/14lz7j5/ntkaware_scaled_rope_allows_llama_models_to_have/
+    # https://github.com/lucidrains/rotary-embedding-torch/blob/main/rotary_embedding_torch/rotary_embedding_torch.py
+    theta *= theta_rescale_factor ** (dim / (dim - 2))
+    freqs = 1.0 / (theta ** (torch.arange(0, dim, 2)[: (dim // 2)].float() / dim))
+    t = torch.arange(end, device=freqs.device)  # type: ignore
+    freqs = torch.outer(t, freqs).float()  # type: ignore
+    freqs_cos = torch.cos(freqs)  # real part
+    freqs_sin = torch.sin(freqs)  # imaginary part
+    return torch.cat([freqs_cos, freqs_sin], dim=-1)
+def load_checkpoint(ckpt_path, use_ema=True):
+    # Load checkpoint based on file extension
+    if ckpt_path.endswith('.safetensors'):
+        print(f"Loading safetensors checkpoint from {ckpt_path}")
+        checkpoint = safetensors.torch.load_file(ckpt_path)
+        # For safetensors, keys are already flattened, check structure
+        if use_ema:
+            # Check if keys contain ema_model_state_dict prefix
+            if any(k.startswith("ema_model_state_dict.") for k in checkpoint.keys()):
+                dict_state = {
+                    k.replace("ema_model_state_dict.", "").replace("ema_model.", ""): v
+                    for k, v in checkpoint.items()
+                    if k.startswith("ema_model_state_dict.") and "initted" not in k and "step" not in k
+                }
+            # Check if keys contain ema_model prefix directly
+            elif any(k.startswith("ema_model.") for k in checkpoint.keys()):
+                dict_state = {
+                    k.replace("ema_model.", ""): v
+                    for k, v in checkpoint.items()
+                    if k.startswith("ema_model.") and "initted" not in k and "step" not in k
+                }
+            else:
+                # Keys are already in the expected format
+                dict_state = checkpoint
+        else:
+            dict_state = checkpoint
+    else:
+        print(f"Loading PyTorch checkpoint from {ckpt_path}")
+        checkpoint = torch.load(ckpt_path, weights_only=True)
+        if use_ema:
+            checkpoint["model_state_dict"] = {
+                k.replace("ema_model.", ""): v
+                for k, v in checkpoint["ema_model_state_dict"].items()
+                if k not in ["initted", "step"]
+            }
+        dict_state = checkpoint["model_state_dict"]
+    text_embed_dict = {}
+    for key in dict_state.keys():
+        # transformer.text_embed.text_embed.weight -> text_embed.weight
+        if "text_embed" in key:
+            text_embed_dict[key.replace("transformer.text_embed.", "")] = dict_state[key]
+    return text_embed_dict
+class F5TTS(object):
+    def __init__(
+        self,
+        config,
+        debug_mode=True,
+        stream: Optional[torch.cuda.Stream] = None,
+        tllm_model_dir: Optional[str] = None,
+        model_path: Optional[str] = None,
+        vocab_size: Optional[int] = None,
+    ):
+        self.dtype = config["pretrained_config"]["dtype"]
+        rank = tensorrt_llm.mpi_rank()
+        world_size = config["pretrained_config"]["mapping"]["world_size"]
+        cp_size = config["pretrained_config"]["mapping"]["cp_size"]
+        tp_size = config["pretrained_config"]["mapping"]["tp_size"]
+        pp_size = config["pretrained_config"]["mapping"]["pp_size"]
+        assert pp_size == 1
+        self.mapping = tensorrt_llm.Mapping(
+            world_size=world_size, rank=rank, cp_size=cp_size, tp_size=tp_size, pp_size=1, gpus_per_node=1
+        )
+        local_rank = rank % self.mapping.gpus_per_node
+        self.device = torch.device(f"cuda:{local_rank}")
+        torch.cuda.set_device(self.device)
+        self.stream = stream
+        if self.stream is None:
+            self.stream = torch.cuda.Stream(self.device)
+        torch.cuda.set_stream(self.stream)
+        engine_file = os.path.join(tllm_model_dir, f"rank{rank}.engine")
+        logger.info(f"Loading engine from {engine_file}")
+        with open(engine_file, "rb") as f:
+            engine_buffer = f.read()
+        assert engine_buffer is not None
+        self.session = Session.from_serialized_engine(engine_buffer)
+        self.debug_mode = debug_mode
+        self.inputs = {}
+        self.outputs = {}
+        self.buffer_allocated = False
+        expected_tensor_names = ["noise", "cond", "time", "rope_cos", "rope_sin", "input_lengths", "denoised"]
+        found_tensor_names = [self.session.engine.get_tensor_name(i) for i in range(self.session.engine.num_io_tensors)]
+        if not self.debug_mode and set(expected_tensor_names) != set(found_tensor_names):
+            logger.error(
+                f"The following expected tensors are not found: {set(expected_tensor_names).difference(set(found_tensor_names))}"
+            )
+            logger.error(
+                f"Those tensors in engine are not expected: {set(found_tensor_names).difference(set(expected_tensor_names))}"
+            )
+            logger.error(f"Expected tensor names: {expected_tensor_names}")
+            logger.error(f"Found tensor names: {found_tensor_names}")
+            raise RuntimeError("Tensor names in engine are not the same as expected.")
+        if self.debug_mode:
+            self.debug_tensors = list(set(found_tensor_names) - set(expected_tensor_names))
+        self.max_mel_len = 4096
+        self.text_embedding = TextEmbedding(
+            text_num_embeds=vocab_size, text_dim=512, conv_layers=4, precompute_max_pos=self.max_mel_len
+        ).to(self.device)
+        self.text_embedding.load_state_dict(load_checkpoint(model_path), strict=True)
+        self.target_audio_sample_rate = 24000
+        self.target_rms = 0.15  # target rms for audio
+        self.n_fft = 1024
+        self.win_length = 1024
+        self.hop_length = 256
+        self.n_mel_channels = 100
+        # self.max_mel_len = 3000
+        self.head_dim = 64
+        self.base_rescale_factor = 1.0
+        self.interpolation_factor = 1.0
+        base = 10000.0 * self.base_rescale_factor ** (self.head_dim / (self.head_dim - 2))
+        inv_freq = 1.0 / (base ** (torch.arange(0, self.head_dim, 2).float() / self.head_dim))
+        freqs = torch.outer(torch.arange(self.max_mel_len, dtype=torch.float32), inv_freq) / self.interpolation_factor
+        self.freqs = freqs.repeat_interleave(2, dim=-1).unsqueeze(0)
+        self.rope_cos = self.freqs.cos().half()
+        self.rope_sin = self.freqs.sin().half()
+        self.nfe_steps = 16
+        t = torch.linspace(0, 1, self.nfe_steps + 1, dtype=torch.float32)
+        time_step = t + (-1.0) * (torch.cos(torch.pi * 0.5 * t) - 1 + t)
+        delta_t = torch.diff(time_step)
+        # WAR: hard coding 256 here
+        tmp_dim = 256
+        time_expand = torch.zeros((1, self.nfe_steps, tmp_dim), dtype=torch.float32)
+        half_dim = tmp_dim // 2
+        emb_factor = math.log(10000) / (half_dim - 1)
+        emb_factor = 1000.0 * torch.exp(torch.arange(half_dim, dtype=torch.float32) * -emb_factor)
+        for i in range(self.nfe_steps):
+            emb = time_step[i] * emb_factor
+            time_expand[:, i, :] = torch.cat((emb.sin(), emb.cos()), dim=-1)
+        self.time_expand = time_expand.to(self.device)
+        self.delta_t = torch.cat((delta_t, delta_t), dim=0).contiguous().to(self.device)
+    def _tensor_dtype(self, name):
+        # return torch dtype given tensor name for convenience
+        dtype = trt_dtype_to_torch(self.session.engine.get_tensor_dtype(name))
+        return dtype
+    def _setup(self, batch_size, seq_len):
+        for i in range(self.session.engine.num_io_tensors):
+            name = self.session.engine.get_tensor_name(i)
+            if self.session.engine.get_tensor_mode(name) == trt.TensorIOMode.OUTPUT:
+                shape = list(self.session.engine.get_tensor_shape(name))
+                shape[0] = batch_size
+                shape[1] = seq_len
+                self.outputs[name] = torch.empty(shape, dtype=self._tensor_dtype(name), device=self.device)
+        self.buffer_allocated = True
+    def cuda_stream_guard(func):
+        """Sync external stream and set current stream to the one bound to the session. Reset on exit."""
+        @wraps(func)
+        def wrapper(self, *args, **kwargs):
+            external_stream = torch.cuda.current_stream()
+            if external_stream != self.stream:
+                external_stream.synchronize()
+                torch.cuda.set_stream(self.stream)
+            ret = func(self, *args, **kwargs)
+            if external_stream != self.stream:
+                self.stream.synchronize()
+                torch.cuda.set_stream(external_stream)
+            return ret
+        return wrapper
+    @cuda_stream_guard
+    def forward(
+        self,
+        noise: torch.Tensor,
+        cond: torch.Tensor,
+        time_expand: torch.Tensor,
+        rope_cos: torch.Tensor,
+        rope_sin: torch.Tensor,
+        input_lengths: torch.Tensor,
+        delta_t: torch.Tensor,
+        use_perf: bool = False,
+    ):
+        if use_perf:
+            torch.cuda.nvtx.range_push("flow matching")
+        cfg_strength = 2.0
+        batch_size = noise.shape[0]
+        half_batch = batch_size // 2
+        noise_half = noise[:half_batch]  # Store the initial half of noise
+        input_type = str_dtype_to_torch(self.dtype)
+        # Keep a copy of the initial tensors
+        cond = cond.to(input_type)
+        rope_cos = rope_cos.to(input_type)
+        rope_sin = rope_sin.to(input_type)
+        input_lengths = input_lengths.to(str_dtype_to_torch("int32"))
+        # Instead of iteratively updating noise within a single model context,
+        # we'll do a single forward pass for each iteration with fresh context setup
+        for i in range(self.nfe_steps):
+            # Re-setup the buffers for clean execution
+            self._setup(batch_size, noise.shape[1])
+            if not self.buffer_allocated:
+                raise RuntimeError("Buffer not allocated, please call setup first!")
+            # Re-create combined noises for this iteration
+            current_noise = torch.cat([noise_half, noise_half], dim=0).to(input_type)
+            # Get time step for this iteration
+            current_time = time_expand[:, i].to(input_type)
+            # Create fresh input dictionary for this iteration
+            current_inputs = {
+                "noise": current_noise,
+                "cond": cond,
+                "time": current_time,
+                "rope_cos": rope_cos,
+                "rope_sin": rope_sin,
+                "input_lengths": input_lengths,
+            }
+            # Update inputs and set shapes
+            self.inputs.clear()  # Clear previous inputs
+            self.inputs.update(**current_inputs)
+            self.session.set_shapes(self.inputs)
+            if use_perf:
+                torch.cuda.nvtx.range_push(f"execute {i}")
+            ok = self.session.run(self.inputs, self.outputs, self.stream.cuda_stream)
+            assert ok, "Failed to execute model"
+            # self.session.context.execute_async_v3(self.stream.cuda_stream)
+            if use_perf:
+                torch.cuda.nvtx.range_pop()
+            # Process results
+            t_scale = delta_t[i].unsqueeze(0).to(input_type)
+            # Extract predictions
+            pred_cond = self.outputs["denoised"][:half_batch]
+            pred_uncond = self.outputs["denoised"][half_batch:]
+            # Apply classifier-free guidance with safeguards
+            guidance = pred_cond + (pred_cond - pred_uncond) * cfg_strength
+            # Calculate update for noise
+            noise_half = noise_half + guidance * t_scale
+        if use_perf:
+            torch.cuda.nvtx.range_pop()
+        return noise_half
+    def sample(
+        self,
+        text_pad_sequence: torch.Tensor,
+        ref_mel_batch: torch.Tensor,
+        ref_mel_len_batch: torch.Tensor,
+        estimated_reference_target_mel_len: List[int],
+        remove_input_padding: bool = False,
+        use_perf: bool = False,
+    ):
+        if use_perf:
+            torch.cuda.nvtx.range_push("text embedding")
+        batch = text_pad_sequence.shape[0]
+        max_seq_len = ref_mel_batch.shape[1]
+        text_pad_sequence_drop = torch.cat(
+            (text_pad_sequence, torch.zeros((1, text_pad_sequence.shape[1]), dtype=torch.int32).to(self.device)), dim=0
+        )
+        text_embedding_drop_list = []
+        for i in range(batch + 1):
+            text_embedding_drop_list.append(self.text_embedding(text_pad_sequence_drop[i].unsqueeze(0).to(self.device)))
+        text_embedding_drop_condition = torch.cat(text_embedding_drop_list, dim=0)
+        text_embedding = text_embedding_drop_condition[:-1]
+        # text_embedding_drop B,T,C batch should be the same
+        text_embedding_drop = text_embedding_drop_condition[-1].unsqueeze(0).repeat(batch, 1, 1)
+        noise = torch.randn_like(ref_mel_batch).to(self.device)
+        rope_cos = self.rope_cos[:, :max_seq_len, :].float().repeat(batch, 1, 1)
+        rope_sin = self.rope_sin[:, :max_seq_len, :].float().repeat(batch, 1, 1)
+        cat_mel_text = torch.cat((ref_mel_batch, text_embedding), dim=-1)
+        cat_mel_text_drop = torch.cat(
+            (
+                torch.zeros((batch, max_seq_len, self.n_mel_channels), dtype=torch.float32).to(self.device),
+                text_embedding_drop,
+            ),
+            dim=-1,
+        )
+        time_expand = self.time_expand.repeat(2 * batch, 1, 1).contiguous()
+        # Convert estimated_reference_target_mel_len to tensor
+        input_lengths = torch.tensor(estimated_reference_target_mel_len, dtype=torch.int32)
+        # combine above along the batch dimension
+        inputs = {
+            "noise": torch.cat((noise, noise), dim=0).contiguous(),
+            "cond": torch.cat((cat_mel_text, cat_mel_text_drop), dim=0).contiguous(),
+            "time_expand": time_expand,
+            "rope_cos": torch.cat((rope_cos, rope_cos), dim=0).contiguous(),
+            "rope_sin": torch.cat((rope_sin, rope_sin), dim=0).contiguous(),
+            "input_lengths": torch.cat((input_lengths, input_lengths), dim=0).contiguous(),
+            "delta_t": self.delta_t,
+        }
+        if use_perf and remove_input_padding:
+            torch.cuda.nvtx.range_push("remove input padding")
+        if remove_input_padding:
+            max_seq_len = inputs["cond"].shape[1]
+            inputs["noise"] = remove_tensor_padding(inputs["noise"], inputs["input_lengths"])
+            inputs["cond"] = remove_tensor_padding(inputs["cond"], inputs["input_lengths"])
+            # for time_expand, convert from B,D to B,T,D by repeat
+            inputs["time_expand"] = inputs["time_expand"].unsqueeze(1).repeat(1, max_seq_len, 1, 1)
+            inputs["time_expand"] = remove_tensor_padding(inputs["time_expand"], inputs["input_lengths"])
+            inputs["rope_cos"] = remove_tensor_padding(inputs["rope_cos"], inputs["input_lengths"])
+            inputs["rope_sin"] = remove_tensor_padding(inputs["rope_sin"], inputs["input_lengths"])
+        if use_perf and remove_input_padding:
+            torch.cuda.nvtx.range_pop()
+        for key in inputs:
+            inputs[key] = inputs[key].to(self.device)
+        if use_perf:
+            torch.cuda.nvtx.range_pop()
+        start_time = time.time()
+        denoised = self.forward(**inputs, use_perf=use_perf)
+        cost_time = time.time() - start_time
+        if use_perf and remove_input_padding:
+            torch.cuda.nvtx.range_push("remove input padding output")
+        if remove_input_padding:
+            denoised_list = []
+            start_idx = 0
+            for i in range(batch):
+                denoised_list.append(denoised[start_idx : start_idx + inputs["input_lengths"][i]])
+                start_idx += inputs["input_lengths"][i]
+            if use_perf and remove_input_padding:
+                torch.cuda.nvtx.range_pop()
+            return denoised_list, cost_time
+        return denoised, cost_time

2flow/services/triton/f5_tts_triton_server/f5_tts/1/model.py ADDED Viewed

	@@ -0,0 +1,278 @@

+# Copyright 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+#
+# Redistribution and use in source and binary forms, with or without
+# modification, are permitted provided that the following conditions
+# are met:
+#  * Redistributions of source code must retain the above copyright
+#    notice, this list of conditions and the following disclaimer.
+#  * Redistributions in binary form must reproduce the above copyright
+#    notice, this list of conditions and the following disclaimer in the
+#    documentation and/or other materials provided with the distribution.
+#  * Neither the name of NVIDIA CORPORATION nor the names of its
+#    contributors may be used to endorse or promote products derived
+#    from this software without specific prior written permission.
+#
+# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
+# EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
+# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
+# PURPOSE ARE DISCLAIMED.  IN NO EVENT SHALL THE COPYRIGHT OWNER OR
+# CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
+# EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
+# PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
+# PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
+# OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
+# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
+# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
+import json
+import os
+import jieba
+import torch
+import torch.nn.functional as F
+import torchaudio
+import triton_python_backend_utils as pb_utils
+from f5_tts_trtllm import F5TTS
+from pypinyin import Style, lazy_pinyin
+from torch.nn.utils.rnn import pad_sequence
+from torch.utils.dlpack import from_dlpack, to_dlpack
+def get_tokenizer(vocab_file_path: str):
+    """
+    tokenizer   - "pinyin" do g2p for only chinese characters, need .txt vocab_file
+                - "char" for char-wise tokenizer, need .txt vocab_file
+                - "byte" for utf-8 tokenizer
+                - "custom" if you're directly passing in a path to the vocab.txt you want to use
+    vocab_size  - if use "pinyin", all available pinyin types, common alphabets (also those with accent) and symbols
+                - if use "char", derived from unfiltered character & symbol counts of custom dataset
+                - if use "byte", set to 256 (unicode byte range)
+    """
+    with open(vocab_file_path, "r", encoding="utf-8") as f:
+        vocab_char_map = {}
+        for i, char in enumerate(f):
+            vocab_char_map[char[:-1]] = i
+    vocab_size = len(vocab_char_map)
+    return vocab_char_map, vocab_size
+def convert_char_to_pinyin(reference_target_texts_list, polyphone=True):
+    final_reference_target_texts_list = []
+    custom_trans = str.maketrans(
+        {";": ",", "“": '"', "”": '"', "‘": "'", "’": "'"}
+    )  # add custom trans here, to address oov
+    def is_chinese(c):
+        return "\u3100" <= c <= "\u9fff"  # common chinese characters
+    for text in reference_target_texts_list:
+        char_list = []
+        text = text.translate(custom_trans)
+        for seg in jieba.cut(text):
+            seg_byte_len = len(bytes(seg, "UTF-8"))
+            if seg_byte_len == len(seg):  # if pure alphabets and symbols
+                if char_list and seg_byte_len > 1 and char_list[-1] not in " :'\"":
+                    char_list.append(" ")
+                char_list.extend(seg)
+            elif polyphone and seg_byte_len == 3 * len(seg):  # if pure east asian characters
+                seg_ = lazy_pinyin(seg, style=Style.TONE3, tone_sandhi=True)
+                for i, c in enumerate(seg):
+                    if is_chinese(c):
+                        char_list.append(" ")
+                    char_list.append(seg_[i])
+            else:  # if mixed characters, alphabets and symbols
+                for c in seg:
+                    if ord(c) < 256:
+                        char_list.extend(c)
+                    elif is_chinese(c):
+                        char_list.append(" ")
+                        char_list.extend(lazy_pinyin(c, style=Style.TONE3, tone_sandhi=True))
+                    else:
+                        char_list.append(c)
+        final_reference_target_texts_list.append(char_list)
+    return final_reference_target_texts_list
+def list_str_to_idx(
+    text: list[str] | list[list[str]],
+    vocab_char_map: dict[str, int],  # {char: idx}
+    padding_value=-1,
+):  # noqa: F722
+    list_idx_tensors = [torch.tensor([vocab_char_map.get(c, 0) for c in t]) for t in text]  # pinyin or char style
+    return list_idx_tensors
+class TritonPythonModel:
+    def initialize(self, args):
+        self.use_perf = True
+        self.device = torch.device("cuda")
+        self.target_audio_sample_rate = 24000
+        self.target_rms = 0.15  # target rms for audio
+        self.n_fft = 1024
+        self.win_length = 1024
+        self.hop_length = 256
+        self.n_mel_channels = 100
+        self.max_mel_len = 3000
+        self.head_dim = 64
+        parameters = json.loads(args["model_config"])["parameters"]
+        for key, value in parameters.items():
+            parameters[key] = value["string_value"]
+        self.vocab_char_map, self.vocab_size = get_tokenizer(parameters["vocab_file"])
+        self.reference_sample_rate = int(parameters["reference_audio_sample_rate"])
+        self.resampler = torchaudio.transforms.Resample(self.reference_sample_rate, self.target_audio_sample_rate)
+        self.tllm_model_dir = parameters["tllm_model_dir"]
+        config_file = os.path.join(self.tllm_model_dir, "config.json")
+        with open(config_file) as f:
+            config = json.load(f)
+        self.model = F5TTS(
+            config,
+            debug_mode=False,
+            tllm_model_dir=self.tllm_model_dir,
+            model_path=parameters["model_path"],
+            vocab_size=self.vocab_size,
+        )
+        self.vocoder = parameters["vocoder"]
+        assert self.vocoder in ["vocos", "bigvgan"]
+        if self.vocoder == "vocos":
+            self.mel_stft = torchaudio.transforms.MelSpectrogram(
+                sample_rate=self.target_audio_sample_rate,
+                n_fft=self.n_fft,
+                win_length=self.win_length,
+                hop_length=self.hop_length,
+                n_mels=self.n_mel_channels,
+                power=1,
+                center=True,
+                normalized=False,
+                norm=None,
+            ).to(self.device)
+            self.compute_mel_fn = self.get_vocos_mel_spectrogram
+        elif self.vocoder == "bigvgan":
+            self.compute_mel_fn = self.get_bigvgan_mel_spectrogram
+    def get_vocos_mel_spectrogram(self, waveform):
+        mel = self.mel_stft(waveform)
+        mel = mel.clamp(min=1e-5).log()
+        return mel.transpose(1, 2)
+    def forward_vocoder(self, mel):
+        mel = mel.to(torch.float32).contiguous().cpu()
+        input_tensor_0 = pb_utils.Tensor.from_dlpack("mel", to_dlpack(mel))
+        inference_request = pb_utils.InferenceRequest(
+            model_name="vocoder", requested_output_names=["waveform"], inputs=[input_tensor_0]
+        )
+        inference_response = inference_request.exec()
+        if inference_response.has_error():
+            raise pb_utils.TritonModelException(inference_response.error().message())
+        else:
+            waveform = pb_utils.get_output_tensor_by_name(inference_response, "waveform")
+            waveform = torch.utils.dlpack.from_dlpack(waveform.to_dlpack()).cpu()
+            return waveform
+    def execute(self, requests):
+        (
+            reference_text_list,
+            target_text_list,
+            reference_target_texts_list,
+            estimated_reference_target_mel_len,
+            reference_mel_len,
+        ) = [], [], [], [], []
+        mel_features_list = []
+        if self.use_perf:
+            torch.cuda.nvtx.range_push("preprocess")
+        for request in requests:
+            wav_tensor = pb_utils.get_input_tensor_by_name(request, "reference_wav")
+            wav_lens = pb_utils.get_input_tensor_by_name(request, "reference_wav_len")
+            reference_text = pb_utils.get_input_tensor_by_name(request, "reference_text").as_numpy()
+            reference_text = reference_text[0][0].decode("utf-8")
+            reference_text_list.append(reference_text)
+            target_text = pb_utils.get_input_tensor_by_name(request, "target_text").as_numpy()
+            target_text = target_text[0][0].decode("utf-8")
+            target_text_list.append(target_text)
+            text = reference_text + target_text
+            reference_target_texts_list.append(text)
+            wav = from_dlpack(wav_tensor.to_dlpack())
+            wav_len = from_dlpack(wav_lens.to_dlpack())
+            wav_len = wav_len.squeeze()
+            assert wav.shape[0] == 1, "Only support batch size 1 for now."
+            wav = wav[:, :wav_len]
+            ref_rms = torch.sqrt(torch.mean(torch.square(wav)))
+            if ref_rms < self.target_rms:
+                wav = wav * self.target_rms / ref_rms
+            if self.reference_sample_rate != self.target_audio_sample_rate:
+                wav = self.resampler(wav)
+            wav = wav.to(self.device)
+            if self.use_perf:
+                torch.cuda.nvtx.range_push("compute_mel")
+            mel_features = self.compute_mel_fn(wav)
+            if self.use_perf:
+                torch.cuda.nvtx.range_pop()
+            mel_features_list.append(mel_features)
+            reference_mel_len.append(mel_features.shape[1])
+            estimated_reference_target_mel_len.append(
+                int(
+                    mel_features.shape[1] * (1 + len(target_text.encode("utf-8")) / len(reference_text.encode("utf-8")))
+                )
+            )
+        max_seq_len = min(max(estimated_reference_target_mel_len), self.max_mel_len)
+        batch = len(requests)
+        mel_features = torch.zeros((batch, max_seq_len, self.n_mel_channels), dtype=torch.float16).to(self.device)
+        for i, mel in enumerate(mel_features_list):
+            mel_features[i, : mel.shape[1], :] = mel
+        reference_mel_len_tensor = torch.LongTensor(reference_mel_len).to(self.device)
+        pinyin_list = convert_char_to_pinyin(reference_target_texts_list, polyphone=True)
+        text_pad_sequence = list_str_to_idx(pinyin_list, self.vocab_char_map)
+        for i, item in enumerate(text_pad_sequence):
+            text_pad_sequence[i] = F.pad(
+                item, (0, estimated_reference_target_mel_len[i] - len(item)), mode="constant", value=-1
+            )
+            text_pad_sequence[i] += 1  # WAR: 0 is reserved for padding token, hard coding in F5-TTS
+        text_pad_sequence = pad_sequence(text_pad_sequence, padding_value=-1, batch_first=True).to(self.device)
+        text_pad_sequence = F.pad(
+            text_pad_sequence, (0, max_seq_len - text_pad_sequence.shape[1]), mode="constant", value=-1
+        )
+        if self.use_perf:
+            torch.cuda.nvtx.range_pop()
+        denoised, cost_time = self.model.sample(
+            text_pad_sequence,
+            mel_features,
+            reference_mel_len_tensor,
+            estimated_reference_target_mel_len,
+            remove_input_padding=False,
+            use_perf=self.use_perf,
+        )
+        if self.use_perf:
+            torch.cuda.nvtx.range_push("vocoder")
+        responses = []
+        for i in range(batch):
+            ref_me_len = reference_mel_len[i]
+            estimated_mel_len = estimated_reference_target_mel_len[i]
+            denoised_one_item = denoised[i, ref_me_len:estimated_mel_len, :].unsqueeze(0).transpose(1, 2)
+            audio = self.forward_vocoder(denoised_one_item)
+            rms = torch.sqrt(torch.mean(torch.square(audio)))
+            if rms < self.target_rms:
+                audio = audio * self.target_rms / rms
+            audio = pb_utils.Tensor.from_dlpack("waveform", to_dlpack(audio))
+            inference_response = pb_utils.InferenceResponse(output_tensors=[audio])
+            responses.append(inference_response)
+        if self.use_perf:
+            torch.cuda.nvtx.range_pop()
+        return responses

2flow/services/triton/f5_tts_triton_server/f5_tts/config.pbtxt ADDED Viewed

	@@ -0,0 +1,81 @@

+# Copyright (c) 2025, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+name: "f5_tts"
+backend: "python"
+max_batch_size: 4
+dynamic_batching {
+    max_queue_delay_microseconds: 1000
+}
+parameters [
+  {
+    key: "vocab_file"
+    value: { string_value: "${vocab}"}
+  },
+  {
+   key: "model_path",
+   value: {string_value:"${model}"}
+  },
+  {
+   key: "tllm_model_dir",
+   value: {string_value:"${trtllm}"}
+  },
+  {
+   key: "reference_audio_sample_rate",
+   value: {string_value:"24000"}
+  },
+  {
+   key: "vocoder",
+   value: {string_value:"${vocoder}"}
+  }
+]
+input [
+  {
+    name: "reference_wav"
+    data_type: TYPE_FP32
+    dims: [-1]
+    optional: True
+  },
+  {
+    name: "reference_wav_len"
+    data_type: TYPE_INT32
+    dims: [1]
+    optional: True
+  },
+  {
+    name: "reference_text"
+    data_type: TYPE_STRING
+    dims: [1]
+  },
+  {
+    name: "target_text"
+    data_type: TYPE_STRING
+    dims: [1]
+  }
+]
+output [
+  {
+    name: "waveform"
+    data_type: TYPE_FP32
+    dims: [ -1 ]
+  }
+]
+instance_group [
+  {
+    count: 1
+    kind: KIND_GPU
+  }
+]

2flow/services/triton/f5_tts_triton_server/vocoder/1/.gitkeep ADDED Viewed

File without changes

2flow/services/triton/f5_tts_triton_server/vocoder/config.pbtxt ADDED Viewed

	@@ -0,0 +1,32 @@

+name: "vocoder"
+backend: "tensorrt"
+default_model_filename: "vocoder.plan"
+max_batch_size: 4
+input [
+  {
+    name: "mel"
+    data_type: TYPE_FP32
+    dims: [ 100, -1 ]
+  }
+]
+output [
+  {
+    name: "waveform"
+    data_type: TYPE_FP32
+    dims: [ -1 ]
+  }
+]
+dynamic_batching {
+    preferred_batch_size: [1, 2, 4]
+    max_queue_delay_microseconds: 1
+}
+instance_group [
+  {
+    count: 1
+    kind: KIND_GPU
+  }
+]

2flow/utils/tts/__pycache__/convert_checkpoint.cpython-310.pyc ADDED Viewed

Binary file (20.5 kB). View file

2flow/utils/tts/__pycache__/convert_checkpoint.cpython-312.pyc ADDED Viewed

Binary file (27.8 kB). View file

2flow/utils/tts/__pycache__/export_vocoder_to_onnx.cpython-312.pyc ADDED Viewed

Binary file (6.35 kB). View file

2flow/utils/tts/convert_checkpoint.py ADDED Viewed

	@@ -0,0 +1,378 @@

+import argparse
+import json
+import os
+import re
+import time
+import traceback
+from concurrent.futures import ThreadPoolExecutor, as_completed
+import safetensors.torch
+import torch
+from tensorrt_llm import str_dtype_to_torch
+from tensorrt_llm.mapping import Mapping
+from tensorrt_llm.models.convert_utils import split, split_matrix_tp
+def split_q_tp(v, n_head, n_hidden, tensor_parallel, rank):
+    split_v = split(v, tensor_parallel, rank, dim=1)
+    return split_v.contiguous()
+def split_q_bias_tp(v, n_head, n_hidden, tensor_parallel, rank):
+    split_v = split(v, tensor_parallel, rank, dim=0)
+    return split_v.contiguous()
+FACEBOOK_DIT_NAME_MAPPING = {
+    "^time_embed.time_mlp.0.weight$": "time_embed.mlp1.weight",
+    "^time_embed.time_mlp.0.bias$": "time_embed.mlp1.bias",
+    "^time_embed.time_mlp.2.weight$": "time_embed.mlp2.weight",
+    "^time_embed.time_mlp.2.bias$": "time_embed.mlp2.bias",
+    "^input_embed.conv_pos_embed.conv1d.0.weight$": "input_embed.conv_pos_embed.conv1d1.weight",
+    "^input_embed.conv_pos_embed.conv1d.0.bias$": "input_embed.conv_pos_embed.conv1d1.bias",
+    "^input_embed.conv_pos_embed.conv1d.2.weight$": "input_embed.conv_pos_embed.conv1d2.weight",
+    "^input_embed.conv_pos_embed.conv1d.2.bias$": "input_embed.conv_pos_embed.conv1d2.bias",
+    "^transformer_blocks.0.attn.to_out.0.weight$": "transformer_blocks.0.attn.to_out.weight",
+    "^transformer_blocks.0.attn.to_out.0.bias$": "transformer_blocks.0.attn.to_out.bias",
+    "^transformer_blocks.1.attn.to_out.0.weight$": "transformer_blocks.1.attn.to_out.weight",
+    "^transformer_blocks.1.attn.to_out.0.bias$": "transformer_blocks.1.attn.to_out.bias",
+    "^transformer_blocks.2.attn.to_out.0.weight$": "transformer_blocks.2.attn.to_out.weight",
+    "^transformer_blocks.2.attn.to_out.0.bias$": "transformer_blocks.2.attn.to_out.bias",
+    "^transformer_blocks.3.attn.to_out.0.weight$": "transformer_blocks.3.attn.to_out.weight",
+    "^transformer_blocks.3.attn.to_out.0.bias$": "transformer_blocks.3.attn.to_out.bias",
+    "^transformer_blocks.4.attn.to_out.0.weight$": "transformer_blocks.4.attn.to_out.weight",
+    "^transformer_blocks.4.attn.to_out.0.bias$": "transformer_blocks.4.attn.to_out.bias",
+    "^transformer_blocks.5.attn.to_out.0.weight$": "transformer_blocks.5.attn.to_out.weight",
+    "^transformer_blocks.5.attn.to_out.0.bias$": "transformer_blocks.5.attn.to_out.bias",
+    "^transformer_blocks.6.attn.to_out.0.weight$": "transformer_blocks.6.attn.to_out.weight",
+    "^transformer_blocks.6.attn.to_out.0.bias$": "transformer_blocks.6.attn.to_out.bias",
+    "^transformer_blocks.7.attn.to_out.0.weight$": "transformer_blocks.7.attn.to_out.weight",
+    "^transformer_blocks.7.attn.to_out.0.bias$": "transformer_blocks.7.attn.to_out.bias",
+    "^transformer_blocks.8.attn.to_out.0.weight$": "transformer_blocks.8.attn.to_out.weight",
+    "^transformer_blocks.8.attn.to_out.0.bias$": "transformer_blocks.8.attn.to_out.bias",
+    "^transformer_blocks.9.attn.to_out.0.weight$": "transformer_blocks.9.attn.to_out.weight",
+    "^transformer_blocks.9.attn.to_out.0.bias$": "transformer_blocks.9.attn.to_out.bias",
+    "^transformer_blocks.10.attn.to_out.0.weight$": "transformer_blocks.10.attn.to_out.weight",
+    "^transformer_blocks.10.attn.to_out.0.bias$": "transformer_blocks.10.attn.to_out.bias",
+    "^transformer_blocks.11.attn.to_out.0.weight$": "transformer_blocks.11.attn.to_out.weight",
+    "^transformer_blocks.11.attn.to_out.0.bias$": "transformer_blocks.11.attn.to_out.bias",
+    "^transformer_blocks.12.attn.to_out.0.weight$": "transformer_blocks.12.attn.to_out.weight",
+    "^transformer_blocks.12.attn.to_out.0.bias$": "transformer_blocks.12.attn.to_out.bias",
+    "^transformer_blocks.13.attn.to_out.0.weight$": "transformer_blocks.13.attn.to_out.weight",
+    "^transformer_blocks.13.attn.to_out.0.bias$": "transformer_blocks.13.attn.to_out.bias",
+    "^transformer_blocks.14.attn.to_out.0.weight$": "transformer_blocks.14.attn.to_out.weight",
+    "^transformer_blocks.14.attn.to_out.0.bias$": "transformer_blocks.14.attn.to_out.bias",
+    "^transformer_blocks.15.attn.to_out.0.weight$": "transformer_blocks.15.attn.to_out.weight",
+    "^transformer_blocks.15.attn.to_out.0.bias$": "transformer_blocks.15.attn.to_out.bias",
+    "^transformer_blocks.16.attn.to_out.0.weight$": "transformer_blocks.16.attn.to_out.weight",
+    "^transformer_blocks.16.attn.to_out.0.bias$": "transformer_blocks.16.attn.to_out.bias",
+    "^transformer_blocks.17.attn.to_out.0.weight$": "transformer_blocks.17.attn.to_out.weight",
+    "^transformer_blocks.17.attn.to_out.0.bias$": "transformer_blocks.17.attn.to_out.bias",
+    "^transformer_blocks.18.attn.to_out.0.weight$": "transformer_blocks.18.attn.to_out.weight",
+    "^transformer_blocks.18.attn.to_out.0.bias$": "transformer_blocks.18.attn.to_out.bias",
+    "^transformer_blocks.19.attn.to_out.0.weight$": "transformer_blocks.19.attn.to_out.weight",
+    "^transformer_blocks.19.attn.to_out.0.bias$": "transformer_blocks.19.attn.to_out.bias",
+    "^transformer_blocks.20.attn.to_out.0.weight$": "transformer_blocks.20.attn.to_out.weight",
+    "^transformer_blocks.20.attn.to_out.0.bias$": "transformer_blocks.20.attn.to_out.bias",
+    "^transformer_blocks.21.attn.to_out.0.weight$": "transformer_blocks.21.attn.to_out.weight",
+    "^transformer_blocks.21.attn.to_out.0.bias$": "transformer_blocks.21.attn.to_out.bias",
+    "^transformer_blocks.0.ff.ff.0.0.weight$": "transformer_blocks.0.ff.project_in.weight",
+    "^transformer_blocks.0.ff.ff.0.0.bias$": "transformer_blocks.0.ff.project_in.bias",
+    "^transformer_blocks.0.ff.ff.2.weight$": "transformer_blocks.0.ff.ff.weight",
+    "^transformer_blocks.0.ff.ff.2.bias$": "transformer_blocks.0.ff.ff.bias",
+    "^transformer_blocks.1.ff.ff.0.0.weight$": "transformer_blocks.1.ff.project_in.weight",
+    "^transformer_blocks.1.ff.ff.0.0.bias$": "transformer_blocks.1.ff.project_in.bias",
+    "^transformer_blocks.1.ff.ff.2.weight$": "transformer_blocks.1.ff.ff.weight",
+    "^transformer_blocks.1.ff.ff.2.bias$": "transformer_blocks.1.ff.ff.bias",
+    "^transformer_blocks.2.ff.ff.0.0.weight$": "transformer_blocks.2.ff.project_in.weight",
+    "^transformer_blocks.2.ff.ff.0.0.bias$": "transformer_blocks.2.ff.project_in.bias",
+    "^transformer_blocks.2.ff.ff.2.weight$": "transformer_blocks.2.ff.ff.weight",
+    "^transformer_blocks.2.ff.ff.2.bias$": "transformer_blocks.2.ff.ff.bias",
+    "^transformer_blocks.3.ff.ff.0.0.weight$": "transformer_blocks.3.ff.project_in.weight",
+    "^transformer_blocks.3.ff.ff.0.0.bias$": "transformer_blocks.3.ff.project_in.bias",
+    "^transformer_blocks.3.ff.ff.2.weight$": "transformer_blocks.3.ff.ff.weight",
+    "^transformer_blocks.3.ff.ff.2.bias$": "transformer_blocks.3.ff.ff.bias",
+    "^transformer_blocks.4.ff.ff.0.0.weight$": "transformer_blocks.4.ff.project_in.weight",
+    "^transformer_blocks.4.ff.ff.0.0.bias$": "transformer_blocks.4.ff.project_in.bias",
+    "^transformer_blocks.4.ff.ff.2.weight$": "transformer_blocks.4.ff.ff.weight",
+    "^transformer_blocks.4.ff.ff.2.bias$": "transformer_blocks.4.ff.ff.bias",
+    "^transformer_blocks.5.ff.ff.0.0.weight$": "transformer_blocks.5.ff.project_in.weight",
+    "^transformer_blocks.5.ff.ff.0.0.bias$": "transformer_blocks.5.ff.project_in.bias",
+    "^transformer_blocks.5.ff.ff.2.weight$": "transformer_blocks.5.ff.ff.weight",
+    "^transformer_blocks.5.ff.ff.2.bias$": "transformer_blocks.5.ff.ff.bias",
+    "^transformer_blocks.6.ff.ff.0.0.weight$": "transformer_blocks.6.ff.project_in.weight",
+    "^transformer_blocks.6.ff.ff.0.0.bias$": "transformer_blocks.6.ff.project_in.bias",
+    "^transformer_blocks.6.ff.ff.2.weight$": "transformer_blocks.6.ff.ff.weight",
+    "^transformer_blocks.6.ff.ff.2.bias$": "transformer_blocks.6.ff.ff.bias",
+    "^transformer_blocks.7.ff.ff.0.0.weight$": "transformer_blocks.7.ff.project_in.weight",
+    "^transformer_blocks.7.ff.ff.0.0.bias$": "transformer_blocks.7.ff.project_in.bias",
+    "^transformer_blocks.7.ff.ff.2.weight$": "transformer_blocks.7.ff.ff.weight",
+    "^transformer_blocks.7.ff.ff.2.bias$": "transformer_blocks.7.ff.ff.bias",
+    "^transformer_blocks.8.ff.ff.0.0.weight$": "transformer_blocks.8.ff.project_in.weight",
+    "^transformer_blocks.8.ff.ff.0.0.bias$": "transformer_blocks.8.ff.project_in.bias",
+    "^transformer_blocks.8.ff.ff.2.weight$": "transformer_blocks.8.ff.ff.weight",
+    "^transformer_blocks.8.ff.ff.2.bias$": "transformer_blocks.8.ff.ff.bias",
+    "^transformer_blocks.9.ff.ff.0.0.weight$": "transformer_blocks.9.ff.project_in.weight",
+    "^transformer_blocks.9.ff.ff.0.0.bias$": "transformer_blocks.9.ff.project_in.bias",
+    "^transformer_blocks.9.ff.ff.2.weight$": "transformer_blocks.9.ff.ff.weight",
+    "^transformer_blocks.9.ff.ff.2.bias$": "transformer_blocks.9.ff.ff.bias",
+    "^transformer_blocks.10.ff.ff.0.0.weight$": "transformer_blocks.10.ff.project_in.weight",
+    "^transformer_blocks.10.ff.ff.0.0.bias$": "transformer_blocks.10.ff.project_in.bias",
+    "^transformer_blocks.10.ff.ff.2.weight$": "transformer_blocks.10.ff.ff.weight",
+    "^transformer_blocks.10.ff.ff.2.bias$": "transformer_blocks.10.ff.ff.bias",
+    "^transformer_blocks.11.ff.ff.0.0.weight$": "transformer_blocks.11.ff.project_in.weight",
+    "^transformer_blocks.11.ff.ff.0.0.bias$": "transformer_blocks.11.ff.project_in.bias",
+    "^transformer_blocks.11.ff.ff.2.weight$": "transformer_blocks.11.ff.ff.weight",
+    "^transformer_blocks.11.ff.ff.2.bias$": "transformer_blocks.11.ff.ff.bias",
+    "^transformer_blocks.12.ff.ff.0.0.weight$": "transformer_blocks.12.ff.project_in.weight",
+    "^transformer_blocks.12.ff.ff.0.0.bias$": "transformer_blocks.12.ff.project_in.bias",
+    "^transformer_blocks.12.ff.ff.2.weight$": "transformer_blocks.12.ff.ff.weight",
+    "^transformer_blocks.12.ff.ff.2.bias$": "transformer_blocks.12.ff.ff.bias",
+    "^transformer_blocks.13.ff.ff.0.0.weight$": "transformer_blocks.13.ff.project_in.weight",
+    "^transformer_blocks.13.ff.ff.0.0.bias$": "transformer_blocks.13.ff.project_in.bias",
+    "^transformer_blocks.13.ff.ff.2.weight$": "transformer_blocks.13.ff.ff.weight",
+    "^transformer_blocks.13.ff.ff.2.bias$": "transformer_blocks.13.ff.ff.bias",
+    "^transformer_blocks.14.ff.ff.0.0.weight$": "transformer_blocks.14.ff.project_in.weight",
+    "^transformer_blocks.14.ff.ff.0.0.bias$": "transformer_blocks.14.ff.project_in.bias",
+    "^transformer_blocks.14.ff.ff.2.weight$": "transformer_blocks.14.ff.ff.weight",
+    "^transformer_blocks.14.ff.ff.2.bias$": "transformer_blocks.14.ff.ff.bias",
+    "^transformer_blocks.15.ff.ff.0.0.weight$": "transformer_blocks.15.ff.project_in.weight",
+    "^transformer_blocks.15.ff.ff.0.0.bias$": "transformer_blocks.15.ff.project_in.bias",
+    "^transformer_blocks.15.ff.ff.2.weight$": "transformer_blocks.15.ff.ff.weight",
+    "^transformer_blocks.15.ff.ff.2.bias$": "transformer_blocks.15.ff.ff.bias",
+    "^transformer_blocks.16.ff.ff.0.0.weight$": "transformer_blocks.16.ff.project_in.weight",
+    "^transformer_blocks.16.ff.ff.0.0.bias$": "transformer_blocks.16.ff.project_in.bias",
+    "^transformer_blocks.16.ff.ff.2.weight$": "transformer_blocks.16.ff.ff.weight",
+    "^transformer_blocks.16.ff.ff.2.bias$": "transformer_blocks.16.ff.ff.bias",
+    "^transformer_blocks.17.ff.ff.0.0.weight$": "transformer_blocks.17.ff.project_in.weight",
+    "^transformer_blocks.17.ff.ff.0.0.bias$": "transformer_blocks.17.ff.project_in.bias",
+    "^transformer_blocks.17.ff.ff.2.weight$": "transformer_blocks.17.ff.ff.weight",
+    "^transformer_blocks.17.ff.ff.2.bias$": "transformer_blocks.17.ff.ff.bias",
+    "^transformer_blocks.18.ff.ff.0.0.weight$": "transformer_blocks.18.ff.project_in.weight",
+    "^transformer_blocks.18.ff.ff.0.0.bias$": "transformer_blocks.18.ff.project_in.bias",
+    "^transformer_blocks.18.ff.ff.2.weight$": "transformer_blocks.18.ff.ff.weight",
+    "^transformer_blocks.18.ff.ff.2.bias$": "transformer_blocks.18.ff.ff.bias",
+    "^transformer_blocks.19.ff.ff.0.0.weight$": "transformer_blocks.19.ff.project_in.weight",
+    "^transformer_blocks.19.ff.ff.0.0.bias$": "transformer_blocks.19.ff.project_in.bias",
+    "^transformer_blocks.19.ff.ff.2.weight$": "transformer_blocks.19.ff.ff.weight",
+    "^transformer_blocks.19.ff.ff.2.bias$": "transformer_blocks.19.ff.ff.bias",
+    "^transformer_blocks.20.ff.ff.0.0.weight$": "transformer_blocks.20.ff.project_in.weight",
+    "^transformer_blocks.20.ff.ff.0.0.bias$": "transformer_blocks.20.ff.project_in.bias",
+    "^transformer_blocks.20.ff.ff.2.weight$": "transformer_blocks.20.ff.ff.weight",
+    "^transformer_blocks.20.ff.ff.2.bias$": "transformer_blocks.20.ff.ff.bias",
+    "^transformer_blocks.21.ff.ff.0.0.weight$": "transformer_blocks.21.ff.project_in.weight",
+    "^transformer_blocks.21.ff.ff.0.0.bias$": "transformer_blocks.21.ff.project_in.bias",
+    "^transformer_blocks.21.ff.ff.2.weight$": "transformer_blocks.21.ff.ff.weight",
+    "^transformer_blocks.21.ff.ff.2.bias$": "transformer_blocks.21.ff.ff.bias",
+}
+def parse_arguments():
+    parser = argparse.ArgumentParser()
+    parser.add_argument(
+        "--model_name",
+        type=str,
+        default="F5TTS_Base",
+        choices=[
+            "F5TTS_Base",
+            "F5TTS_v1_Base",
+        ],
+    )  # TODO: support F5TTS_v1_Base
+    parser.add_argument("--timm_ckpt", type=str, default="./ckpts/model_1200000.pt")
+    parser.add_argument(
+        "--output_dir", type=str, default="./tllm_checkpoint", help="The path to save the TensorRT-LLM checkpoint"
+    )
+    parser.add_argument("--hidden_size", type=int, default=1024, help="The hidden size of DiT")
+    parser.add_argument("--depth", type=int, default=22, help="The number of DiTBlock layers")
+    parser.add_argument("--num_heads", type=int, default=16, help="The number of heads of attention module")
+    parser.add_argument("--cfg_scale", type=float, default=4.0)
+    parser.add_argument("--tp_size", type=int, default=1, help="N-way tensor parallelism size")
+    parser.add_argument("--cp_size", type=int, default=1, help="Context parallelism size")
+    parser.add_argument("--pp_size", type=int, default=1, help="N-way pipeline parallelism size")
+    parser.add_argument("--dtype", type=str, default="float16", choices=["float32", "bfloat16", "float16"])
+    parser.add_argument("--fp8_linear", action="store_true", help="Whether use FP8 for linear layers")
+    parser.add_argument(
+        "--workers", type=int, default=1, help="The number of workers for converting checkpoint in parallel"
+    )
+    args = parser.parse_args()
+    return args
+def convert_timm_dit(args, mapping, dtype="float32"):
+    weights = {}
+    tik = time.time()
+    torch_dtype = str_dtype_to_torch(dtype)
+    tensor_parallel = mapping.tp_size
+    # Load checkpoint based on file extension
+    if args.timm_ckpt.endswith('.safetensors'):
+        print(f"Loading safetensors checkpoint from {args.timm_ckpt}")
+        model_params = safetensors.torch.load_file(args.timm_ckpt)
+        # For safetensors, check if we need to extract from a nested dict
+        if any(k.startswith("ema_model.transformer") for k in model_params.keys()):
+            model_params = {
+                k: v for k, v in model_params.items() if k.startswith("ema_model.transformer")
+            }
+        elif any(k.startswith("ema_model_state_dict.ema_model.transformer") for k in model_params.keys()):
+            model_params = {
+                k.replace("ema_model_state_dict.", ""): v
+                for k, v in model_params.items()
+                if k.startswith("ema_model_state_dict.ema_model.transformer")
+            }
+    else:
+        print(f"Loading PyTorch checkpoint from {args.timm_ckpt}")
+        checkpoint = torch.load(args.timm_ckpt)
+        model_params = dict(checkpoint)
+        model_params = {
+            k: v for k, v in model_params["ema_model_state_dict"].items() if k.startswith("ema_model.transformer")
+        }
+    prefix = "ema_model.transformer."
+    model_params = {key[len(prefix) :] if key.startswith(prefix) else key: value for key, value in model_params.items()}
+    timm_to_trtllm_name = FACEBOOK_DIT_NAME_MAPPING
+    def get_trtllm_name(timm_name):
+        for k, v in timm_to_trtllm_name.items():
+            m = re.match(k, timm_name)
+            if m is not None:
+                if "*" in v:
+                    v = v.replace("*", m.groups()[0])
+                return v
+        return timm_name
+    weights = dict()
+    for name, param in model_params.items():
+        if name == "input_embed.conv_pos_embed.conv1d.0.weight" or name == "input_embed.conv_pos_embed.conv1d.2.weight":
+            weights[get_trtllm_name(name)] = param.contiguous().to(torch_dtype).unsqueeze(-1)
+        else:
+            weights[get_trtllm_name(name)] = param.contiguous().to(torch_dtype)
+    assert len(weights) == len(model_params)
+    # new_prefix = 'f5_transformer.'
+    new_prefix = ""
+    weights = {new_prefix + key: value for key, value in weights.items()}
+    import math
+    scale_factor = math.pow(64, -0.25)
+    for k, v in weights.items():
+        if re.match("^transformer_blocks.*.attn.to_k.weight$", k):
+            weights[k] *= scale_factor
+            weights[k] = split_q_tp(v, args.num_heads, args.hidden_size, tensor_parallel, mapping.tp_rank)
+        elif re.match("^transformer_blocks.*.attn.to_k.bias$", k):
+            weights[k] *= scale_factor
+            weights[k] = split_q_bias_tp(v, args.num_heads, args.hidden_size, tensor_parallel, mapping.tp_rank)
+        elif re.match("^transformer_blocks.*.attn.to_q.weight$", k):
+            weights[k] = split_q_tp(v, args.num_heads, args.hidden_size, tensor_parallel, mapping.tp_rank)
+            weights[k] *= scale_factor
+        elif re.match("^transformer_blocks.*.attn.to_q.bias$", k):
+            weights[k] = split_q_bias_tp(v, args.num_heads, args.hidden_size, tensor_parallel, mapping.tp_rank)
+            weights[k] *= scale_factor
+        elif re.match("^transformer_blocks.*.attn.to_v.weight$", k):
+            weights[k] = split_q_tp(v, args.num_heads, args.hidden_size, tensor_parallel, mapping.tp_rank)
+        elif re.match("^transformer_blocks.*.attn.to_v.bias$", k):
+            weights[k] = split_q_bias_tp(v, args.num_heads, args.hidden_size, tensor_parallel, mapping.tp_rank)
+        elif re.match("^transformer_blocks.*.attn.to_out.weight$", k):
+            weights[k] = split_matrix_tp(v, tensor_parallel, mapping.tp_rank, dim=1)
+    tok = time.time()
+    t = time.strftime("%H:%M:%S", time.gmtime(tok - tik))
+    print(f"Weights loaded. Total time: {t}")
+    return weights
+def save_config(args):
+    if not os.path.exists(args.output_dir):
+        os.makedirs(args.output_dir)
+    config = {
+        "architecture": "F5TTS",
+        "dtype": args.dtype,
+        "hidden_size": 1024,
+        "num_hidden_layers": 22,
+        "num_attention_heads": 16,
+        "dim_head": 64,
+        "dropout": 0.1,
+        "ff_mult": 2,
+        "mel_dim": 100,
+        "text_num_embeds": 256,
+        "text_dim": 512,
+        "conv_layers": 4,
+        "long_skip_connection": False,
+        "mapping": {
+            "world_size": args.cp_size * args.tp_size * args.pp_size,
+            "cp_size": args.cp_size,
+            "tp_size": args.tp_size,
+            "pp_size": args.pp_size,
+        },
+    }
+    if args.fp8_linear:
+        config["quantization"] = {
+            "quant_algo": "FP8",
+            # TODO: add support for exclude modules.
+            # 'exclude_modules': "*final_layer*",
+        }
+    with open(os.path.join(args.output_dir, "config.json"), "w") as f:
+        json.dump(config, f, indent=4)
+def covert_and_save(args, rank):
+    if rank == 0:
+        save_config(args)
+    mapping = Mapping(
+        world_size=args.cp_size * args.tp_size * args.pp_size,
+        rank=rank,
+        cp_size=args.cp_size,
+        tp_size=args.tp_size,
+        pp_size=args.pp_size,
+    )
+    weights = convert_timm_dit(args, mapping, dtype=args.dtype)
+    safetensors.torch.save_file(weights, os.path.join(args.output_dir, f"rank{rank}.safetensors"))
+def execute(workers, func, args):
+    if workers == 1:
+        for rank, f in enumerate(func):
+            f(args, rank)
+    else:
+        with ThreadPoolExecutor(max_workers=workers) as p:
+            futures = [p.submit(f, args, rank) for rank, f in enumerate(func)]
+            exceptions = []
+            for future in as_completed(futures):
+                try:
+                    future.result()
+                except Exception as e:
+                    traceback.print_exc()
+                    exceptions.append(e)
+            assert len(exceptions) == 0, "Checkpoint conversion failed, please check error log."
+def main():
+    args = parse_arguments()
+    world_size = args.cp_size * args.tp_size * args.pp_size
+    assert args.pp_size == 1, "PP is not supported yet."
+    tik = time.time()
+    if args.timm_ckpt is None:
+        return
+    print("start execute")
+    execute(args.workers, [covert_and_save] * world_size, args)
+    tok = time.time()
+    t = time.strftime("%H:%M:%S", time.gmtime(tok - tik))
+    print(f"Total time of converting checkpoints: {t}")
+if __name__ == "__main__":
+    main()

2flow/utils/tts/export_vocoder_to_onnx.py ADDED Viewed

	@@ -0,0 +1,138 @@

+# Copyright (c) 2024, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import argparse
+import torch
+import torch.nn as nn
+from conv_stft import STFT
+from huggingface_hub import hf_hub_download
+from vocos import Vocos
+opset_version = 17
+def get_args():
+    parser = argparse.ArgumentParser(formatter_class=argparse.ArgumentDefaultsHelpFormatter)
+    parser.add_argument(
+        "--vocoder",
+        type=str,
+        default="vocos",
+        choices=["vocos", "bigvgan"],
+        help="Vocoder to export",
+    )
+    parser.add_argument(
+        "--output-path",
+        type=str,
+        default="./vocos_vocoder.onnx",
+        help="Output path",
+    )
+    return parser.parse_args()
+class ISTFTHead(nn.Module):
+    def __init__(self, n_fft: int, hop_length: int):
+        super().__init__()
+        self.out = None
+        self.stft = STFT(fft_len=n_fft, win_hop=hop_length, win_len=n_fft)
+    def forward(self, x: torch.Tensor):
+        x = self.out(x).transpose(1, 2)
+        mag, p = x.chunk(2, dim=1)
+        mag = torch.exp(mag)
+        mag = torch.clip(mag, max=1e2)
+        real = mag * torch.cos(p)
+        imag = mag * torch.sin(p)
+        audio = self.stft.inverse(input1=real, input2=imag, input_type="realimag")
+        return audio
+class VocosVocoder(nn.Module):
+    def __init__(self, vocos_vocoder):
+        super(VocosVocoder, self).__init__()
+        self.vocos_vocoder = vocos_vocoder
+        istft_head_out = self.vocos_vocoder.head.out
+        n_fft = self.vocos_vocoder.head.istft.n_fft
+        hop_length = self.vocos_vocoder.head.istft.hop_length
+        istft_head_for_export = ISTFTHead(n_fft, hop_length)
+        istft_head_for_export.out = istft_head_out
+        self.vocos_vocoder.head = istft_head_for_export
+    def forward(self, mel):
+        waveform = self.vocos_vocoder.decode(mel)
+        return waveform
+def export_VocosVocoder(vocos_vocoder, output_path, verbose):
+    vocos_vocoder = VocosVocoder(vocos_vocoder).cuda()
+    vocos_vocoder.eval()
+    dummy_batch_size = 8
+    dummy_input_length = 500
+    dummy_mel = torch.randn(dummy_batch_size, 100, dummy_input_length).cuda()
+    with torch.no_grad():
+        dummy_waveform = vocos_vocoder(mel=dummy_mel)
+        print(dummy_waveform.shape)
+    dummy_input = dummy_mel
+    torch.onnx.export(
+        vocos_vocoder,
+        dummy_input,
+        output_path,
+        opset_version=opset_version,
+        do_constant_folding=True,
+        input_names=["mel"],
+        output_names=["waveform"],
+        dynamic_axes={
+            "mel": {0: "batch_size", 2: "input_length"},
+            "waveform": {0: "batch_size", 1: "output_length"},
+        },
+        verbose=verbose,
+    )
+    print("Exported to {}".format(output_path))
+def load_vocoder(vocoder_name="vocos", is_local=False, local_path="", device="cpu", hf_cache_dir=None):
+    if vocoder_name == "vocos":
+        # vocoder = Vocos.from_pretrained("charactr/vocos-mel-24khz").to(device)
+        if is_local:
+            print(f"Load vocos from local path {local_path}")
+            config_path = f"{local_path}/config.yaml"
+            model_path = f"{local_path}/pytorch_model.bin"
+        else:
+            print("Download Vocos from huggingface charactr/vocos-mel-24khz")
+            repo_id = "charactr/vocos-mel-24khz"
+            config_path = hf_hub_download(repo_id=repo_id, cache_dir=hf_cache_dir, filename="config.yaml")
+            model_path = hf_hub_download(repo_id=repo_id, cache_dir=hf_cache_dir, filename="pytorch_model.bin")
+        vocoder = Vocos.from_hparams(config_path)
+        state_dict = torch.load(model_path, map_location="cpu", weights_only=True)
+        vocoder.load_state_dict(state_dict)
+        vocoder = vocoder.eval().to(device)
+    elif vocoder_name == "bigvgan":
+        raise NotImplementedError("BigVGAN is not supported yet")
+        vocoder.remove_weight_norm()
+        vocoder = vocoder.eval().to(device)
+    return vocoder
+if __name__ == "__main__":
+    args = get_args()
+    vocoder = load_vocoder(vocoder_name=args.vocoder, device="cpu", hf_cache_dir=None)
+    if args.vocoder == "vocos":
+        export_VocosVocoder(vocoder, args.output_path, verbose=False)