ONNX
Safetensors
Chinese
English
Dubbing-model
jetjodh commited on
Commit
cbb8be3
·
verified ·
1 Parent(s): 4cc14a3

Upload 41 files

Browse files
Files changed (42) hide show
  1. .gitattributes +1 -0
  2. MelBandRoformer.ckpt +3 -0
  3. Qwen2-0.5B-CosyVoice-BlankEN/LICENSE +202 -0
  4. Qwen2-0.5B-CosyVoice-BlankEN/README.md +94 -0
  5. Qwen2-0.5B-CosyVoice-BlankEN/config.json +27 -0
  6. Qwen2-0.5B-CosyVoice-BlankEN/generation_config.json +14 -0
  7. Qwen2-0.5B-CosyVoice-BlankEN/merges.txt +0 -0
  8. Qwen2-0.5B-CosyVoice-BlankEN/model.safetensors +3 -0
  9. Qwen2-0.5B-CosyVoice-BlankEN/tokenizer_config.json +40 -0
  10. Qwen2-0.5B-CosyVoice-BlankEN/vocab.json +0 -0
  11. README.md +163 -0
  12. asd.onnx +3 -0
  13. config.json +1 -0
  14. face_recog_ir101.onnx +3 -0
  15. fqa.onnx +3 -0
  16. fun_2d.pth +3 -0
  17. fun_2d.zip +3 -0
  18. funcineforge_zh_en/camplus.onnx +3 -0
  19. funcineforge_zh_en/flow/config.yaml +105 -0
  20. funcineforge_zh_en/flow/ds-model.pt.best/mp_rank_00_model_states.pt +3 -0
  21. funcineforge_zh_en/llm/config.yaml +112 -0
  22. funcineforge_zh_en/llm/ds-model.pt.best/mp_rank_00_model_states.pt +3 -0
  23. funcineforge_zh_en/vocoder/config.yaml +26 -0
  24. funcineforge_zh_en/vocoder/ds-model.pt.best/avg_5_removewn.pt +3 -0
  25. funcineforge_zh_en/vocoder/hift_causal.hyper.yaml +38 -0
  26. speech_campplus/.mdl +0 -0
  27. speech_campplus/.msc +0 -0
  28. speech_campplus/.mv +1 -0
  29. speech_campplus/campplus_cn_en_common.pt +3 -0
  30. speech_campplus/config.yaml +23 -0
  31. speech_campplus/configuration.json +23 -0
  32. speech_campplus/structure.png +3 -0
  33. speech_fsmn_vad/.mdl +0 -0
  34. speech_fsmn_vad/.msc +0 -0
  35. speech_fsmn_vad/.mv +1 -0
  36. speech_fsmn_vad/am.mvn +8 -0
  37. speech_fsmn_vad/config.yaml +56 -0
  38. speech_fsmn_vad/configuration.json +13 -0
  39. speech_fsmn_vad/model.pt +3 -0
  40. speech_fsmn_vad/struct.png +0 -0
  41. speech_tokenizer_v3.onnx +3 -0
  42. version-RFB-320.onnx +3 -0
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ speech_campplus/structure.png filter=lfs diff=lfs merge=lfs -text
MelBandRoformer.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:87201f4d31afb5bc79993230fc49446918425574db48c01c405e44f365c7559e
3
+ size 913106900
Qwen2-0.5B-CosyVoice-BlankEN/LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2024 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
Qwen2-0.5B-CosyVoice-BlankEN/README.md ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ language:
4
+ - en
5
+ pipeline_tag: text-generation
6
+ tags:
7
+ - chat
8
+ base_model: Qwen/Qwen2-0.5B
9
+ ---
10
+
11
+ # Qwen2-0.5B-Instruct
12
+
13
+ ## Introduction
14
+
15
+ Qwen2 is the new series of Qwen large language models. For Qwen2, we release a number of base language models and instruction-tuned language models ranging from 0.5 to 72 billion parameters, including a Mixture-of-Experts model. This repo contains the instruction-tuned 0.5B Qwen2 model.
16
+
17
+ Compared with the state-of-the-art opensource language models, including the previous released Qwen1.5, Qwen2 has generally surpassed most opensource models and demonstrated competitiveness against proprietary models across a series of benchmarks targeting for language understanding, language generation, multilingual capability, coding, mathematics, reasoning, etc.
18
+
19
+ For more details, please refer to our [blog](https://qwenlm.github.io/blog/qwen2/), [GitHub](https://github.com/QwenLM/Qwen2), and [Documentation](https://qwen.readthedocs.io/en/latest/).
20
+ <br>
21
+
22
+ ## Model Details
23
+ Qwen2 is a language model series including decoder language models of different model sizes. For each size, we release the base language model and the aligned chat model. It is based on the Transformer architecture with SwiGLU activation, attention QKV bias, group query attention, etc. Additionally, we have an improved tokenizer adaptive to multiple natural languages and codes.
24
+
25
+ ## Training details
26
+ We pretrained the models with a large amount of data, and we post-trained the models with both supervised finetuning and direct preference optimization.
27
+
28
+
29
+ ## Requirements
30
+ The code of Qwen2 has been in the latest Hugging face transformers and we advise you to install `transformers>=4.37.0`, or you might encounter the following error:
31
+ ```
32
+ KeyError: 'qwen2'
33
+ ```
34
+
35
+ ## Quickstart
36
+
37
+ Here provides a code snippet with `apply_chat_template` to show you how to load the tokenizer and model and how to generate contents.
38
+
39
+ ```python
40
+ from transformers import AutoModelForCausalLM, AutoTokenizer
41
+ device = "cuda" # the device to load the model onto
42
+
43
+ model = AutoModelForCausalLM.from_pretrained(
44
+ "Qwen/Qwen2-0.5B-Instruct",
45
+ torch_dtype="auto",
46
+ device_map="auto"
47
+ )
48
+ tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen2-0.5B-Instruct")
49
+
50
+ prompt = "Give me a short introduction to large language model."
51
+ messages = [
52
+ {"role": "system", "content": "You are a helpful assistant."},
53
+ {"role": "user", "content": prompt}
54
+ ]
55
+ text = tokenizer.apply_chat_template(
56
+ messages,
57
+ tokenize=False,
58
+ add_generation_prompt=True
59
+ )
60
+ model_inputs = tokenizer([text], return_tensors="pt").to(device)
61
+
62
+ generated_ids = model.generate(
63
+ model_inputs.input_ids,
64
+ max_new_tokens=512
65
+ )
66
+ generated_ids = [
67
+ output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids)
68
+ ]
69
+
70
+ response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0]
71
+ ```
72
+
73
+ ## Evaluation
74
+
75
+ We briefly compare Qwen2-0.5B-Instruct with Qwen1.5-0.5B-Chat. The results are as follows:
76
+
77
+ | Datasets | Qwen1.5-0.5B-Chat | **Qwen2-0.5B-Instruct** | Qwen1.5-1.8B-Chat | **Qwen2-1.5B-Instruct** |
78
+ | :--- | :---: | :---: | :---: | :---: |
79
+ | MMLU | 35.0 | **37.9** | 43.7 | **52.4** |
80
+ | HumanEval | 9.1 | **17.1** | 25.0 | **37.8** |
81
+ | GSM8K | 11.3 | **40.1** | 35.3 | **61.6** |
82
+ | C-Eval | 37.2 | **45.2** | 55.3 | **63.8** |
83
+ | IFEval (Prompt Strict-Acc.) | 14.6 | **20.0** | 16.8 | **29.0** |
84
+
85
+ ## Citation
86
+
87
+ If you find our work helpful, feel free to give us a cite.
88
+
89
+ ```
90
+ @article{qwen2,
91
+ title={Qwen2 Technical Report},
92
+ year={2024}
93
+ }
94
+ ```
Qwen2-0.5B-CosyVoice-BlankEN/config.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 151643,
7
+ "eos_token_id": 151645,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 896,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 4864,
12
+ "max_position_embeddings": 32768,
13
+ "max_window_layers": 24,
14
+ "model_type": "qwen2",
15
+ "num_attention_heads": 14,
16
+ "num_hidden_layers": 24,
17
+ "num_key_value_heads": 2,
18
+ "rms_norm_eps": 1e-06,
19
+ "rope_theta": 1000000.0,
20
+ "sliding_window": 32768,
21
+ "tie_word_embeddings": true,
22
+ "torch_dtype": "bfloat16",
23
+ "transformers_version": "4.40.1",
24
+ "use_cache": true,
25
+ "use_sliding_window": false,
26
+ "vocab_size": 151936
27
+ }
Qwen2-0.5B-CosyVoice-BlankEN/generation_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "pad_token_id": 151643,
4
+ "do_sample": true,
5
+ "eos_token_id": [
6
+ 151645,
7
+ 151643
8
+ ],
9
+ "repetition_penalty": 1.1,
10
+ "temperature": 0.7,
11
+ "top_p": 0.8,
12
+ "top_k": 20,
13
+ "transformers_version": "4.37.0"
14
+ }
Qwen2-0.5B-CosyVoice-BlankEN/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
Qwen2-0.5B-CosyVoice-BlankEN/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:130282af0dfa9fe5840737cc49a0d339d06075f83c5a315c3372c9a0740d0b96
3
+ size 988097824
Qwen2-0.5B-CosyVoice-BlankEN/tokenizer_config.json ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "151643": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "151644": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "151645": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ }
28
+ },
29
+ "additional_special_tokens": ["<|im_start|>", "<|im_end|>"],
30
+ "bos_token": null,
31
+ "chat_template": "{% for message in messages %}{% if loop.first and messages[0]['role'] != 'system' %}{{ '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}{% endif %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
32
+ "clean_up_tokenization_spaces": false,
33
+ "eos_token": "<|im_end|>",
34
+ "errors": "replace",
35
+ "model_max_length": 32768,
36
+ "pad_token": "<|endoftext|>",
37
+ "split_special_tokens": false,
38
+ "tokenizer_class": "Qwen2Tokenizer",
39
+ "unk_token": null
40
+ }
Qwen2-0.5B-CosyVoice-BlankEN/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
README.md ADDED
@@ -0,0 +1,163 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ datasets:
4
+ - FunAudioLLM/CineDub-Example
5
+ language:
6
+ - zh
7
+ - en
8
+ tags:
9
+ - Dubbing-model
10
+ ---
11
+
12
+ <p align="center">
13
+ <b>🎬 Fun-CineForge: A Unified Dataset Pipeline and Model for Zero-Shot Movie Dubbing in Diverse Cinematic Scenes</b>
14
+ </p>
15
+
16
+ <div align="center">
17
+
18
+ ![license](https://img.shields.io/github/license/modelscope/modelscope.svg)
19
+ <a href=""><img src="https://img.shields.io/badge/OS-Linux-orange.svg"></a>
20
+ <a href=""><img src="https://img.shields.io/badge/Python->=3.8-aff.svg"></a>
21
+ <a href=""><img src="https://img.shields.io/badge/Pytorch->=2.1-blue"></a>
22
+ </div>
23
+
24
+ <div align="center">
25
+ <h4><a href="#Open-Source">Open Source</a>
26
+ |<a href="#Dataset-Pipeline">Dataset Pipeline</a>
27
+ |<a href="#Dubbing-Model">Dubbing Model</a>
28
+ |<a href="#Recent-Updates">Recent Updates</a>
29
+ |<a href="#Publication">Publication</a>
30
+ |<a href="#Comminicate">Comminicate</a>
31
+ </h4>
32
+ </div>
33
+
34
+ **Fun-CineForge** contains an end-to-end dataset pipeline for producing large-scale dubbing datasets and an MLLM-based dubbing model designed for diverse cinematic scenes.
35
+ Using this pipeline, we constructed the first large-scale Chinese television dubbing dataset CineDub-CN, which includes rich annotations and diverse scenes.
36
+ In monologue, narration, dialogue, and multi-speaker scenes, our dubbing model consistently outperforms state-of-the-art methods in terms of audio quality, lip-sync, timbre transition, and instruction following.
37
+
38
+ <a name="Open-Source"></a>
39
+ ## Open Source 🎬
40
+ You can access [https://funcineforge.github.io/](https://funcineforge.github.io/) to get our CineDub-CN dataset samples and demo samples.
41
+
42
+ GitHub link: [https://github.com/FunAudioLLM/FunCineForge/](https://github.com/FunAudioLLM/FunCineForge/)
43
+
44
+ Modelscope link: [https://www.modelscope.cn/models/FunAudioLLM/Fun-CineForge/](https://www.modelscope.cn/models/FunAudioLLM/Fun-CineForge/)
45
+
46
+ CineDub Samples:
47
+ [huggingface](https://huggingface.co/datasets/FunAudioLLM/CineDub-Example/)
48
+ [modelscope](https://www.modelscope.cn/datasets/FunAudioLLM/CineDub-Example)
49
+
50
+ <a name="Dataset-Pipeline"></a>
51
+ ## Dataset Pipeline 🔨
52
+
53
+ ### Environmental Installation
54
+
55
+ Fun-CineForge dataset pipeline toolkit only relies on a Python environment to run.
56
+ ```shell
57
+ # Conda
58
+ git clone git@github.com:FunAudioLLM/FunCineForge.git
59
+ conda create -n FunCineForge python=3.10 -y && conda activate FunCineForge
60
+ sudo apt-get install ffmpeg
61
+ # Initial settings
62
+ python setup.py
63
+ ```
64
+
65
+ ### Data collection
66
+ If you want to produce your own data,
67
+ we recommend that you refer to the following requirements to collect the corresponding movies or television series.
68
+
69
+ 1. Video source: TV dramas or movies, non documentaries, with more monologues or dialogue scenes, clear and unobstructed faces (such as without masks and veils).
70
+ 2. Speech Requirements: Standard pronunciation, clear articulation, prominent human voice. Avoid materials with strong dialects, excessive background noise, or strong colloquialism.
71
+ 3. Image Requirements: High resolution, clear facial details, sufficient lighting, avoiding extremely dark or strong backlit scenes.
72
+
73
+ ### How to use
74
+
75
+ - [1] Standardize video format and name; trim the beginning and end of long videos; extract the audio from the trimmed video. (default is to trim 10 seconds from both the beginning and end.)
76
+ ```shell
77
+ python normalize_trim.py --root datasets/raw_zh --intro 10 --outro 10
78
+ ```
79
+
80
+ - [2] [Speech Separation](./speech_separation/README.md). The audio is used to separate the vocals from the instrumental music.
81
+ ```shell
82
+ cd speech_separation
83
+ python run.py --root datasets/clean/zh --gpus 0 1 2 3
84
+ ```
85
+
86
+ - [3] [VideoClipper](./video_clip/README.md). For long videos, VideoClipper is used to obtain sentence-level subtitle files and clip the long video into segments based on timestamps. Now it supports bilingualism in both Chinese and English. Below is an example in Chinese. It is recommended to use gpu acceleration for English.
87
+ ```shell
88
+ cd video_clip
89
+ bash run.sh --stage 1 --stop_stage 2 --input datasets/raw_zh --output datasets/clean/zh --lang zh --device cpu
90
+ ```
91
+
92
+ - Video duration limit and check for cleanup. (Without --execute, only pre-deleted files will be printed. After checking, add --execute to confirm the deletion.)
93
+ ```shell
94
+ python clean_video.py --root datasets/clean/zh
95
+ python clean_srt.py --root datasets/clean/zh --lang zh
96
+ ```
97
+
98
+ - [4] [Speaker Diarization](./speaker_diarization/README.md). Multimodal active speaker recognition obtains RTTM files; identifies the speaker's facial frames, extracts frame-level speaker face and lip raw data.
99
+ ```shell
100
+ cd speaker_diarization
101
+ bash run.sh --stage 1 --stop_stage 4 --hf_access_token hf_xxx --root datasets/clean/zh --gpus "0 1 2 3"
102
+ ```
103
+
104
+ - [5] Multimodal CoT Correction. Based on general-purpose MLLMs, the system uses audio, ASR text, and RTTM files as input. It leverages Chain-of-Thought (CoT) reasoning to extract clues and corrects the results of the specialized models. It also annotates character age, gender, and vocal timbre. Experimental results show that this strategy reduces the CER from 4.53% to 0.94% and the speaker diarization error rate from 8.38% to 1.20%, achieving quality comparable to or even better than manual transcription. Adding the --resume enables breakpoint COT inference to prevent wasted resources from repeated COT inferences. Now supports both Chinese and English.
105
+ ```shell
106
+ python cot.py --root_dir datasets/clean/zh --lang zh --provider google --model gemini-3-pro-preview --api_key xxx --resume
107
+ python cot.py --root_dir datasets/clean/en --lang en --provider google --model gemini-3-pro-preview --api_key xxx --resume
108
+ python build_datasets.py --root_zh datasets/clean/zh --root_en datasets/clean/en --out_dir datasets/clean --save
109
+ ```
110
+
111
+ - (Reference) Extract speech tokens based on the CosyVoice3 tokenizer for llm training.
112
+ ```shell
113
+ python speech_tokenizer.py --root datasets/clean/zh
114
+ ```
115
+
116
+ <a name="Dubbing-Model"></a>
117
+ ## Dubbing Model ⚙️
118
+ We've open-sourced the inference code and the **infer.sh** script, and provided some test cases in the data folder for your experience. Inference requires a consumer-grade GPU. Run the following command:
119
+
120
+ ```shell
121
+ cd exps
122
+ bash infer.sh
123
+ ```
124
+
125
+ The API for multi-speaker dubbing from raw videos and SRT scripts is under development ...
126
+
127
+ <a name="Recent-Updates"></a>
128
+ ## Recent Updates 🚀
129
+ - 2025/12/18: Fun-CineForge dataset pipeline toolkit is online! 🔥
130
+ - 2026/01/19: Chinese demo samples and CineDub-CN dataset samples released. 🔥
131
+ - 2026/01/25: Fix some environmental and operational issues.
132
+ - 2026/02/09: Optimized the data pipeline and added support for English videos.
133
+ - 2026/03/05: English demo samples and CineDub-EN dataset samples released. 🔥
134
+ - 2026/03/16: Open source inference code and checkpoints. 🔥
135
+
136
+ <a name="Publication"></a>
137
+ ## Publication 📚
138
+ If you use our dataset or code, please cite the following paper:
139
+ <pre>
140
+ @misc{liu2026funcineforgeunifieddatasettoolkit,
141
+ title={FunCineForge: A Unified Dataset Toolkit and Model for Zero-Shot Movie Dubbing in Diverse Cinematic Scenes},
142
+ author={Jiaxuan Liu and Yang Xiang and Han Zhao and Xiangang Li and Zhenhua Ling},
143
+ year={2026},
144
+ eprint={2601.14777},
145
+ archivePrefix={arXiv},
146
+ primaryClass={cs.CV},
147
+ }
148
+ </pre>
149
+
150
+ <a name="Comminicate"></a>
151
+ ## Comminicate 🍟
152
+ We welcome you to participate in discussions on Fun-CineForge [GitHub Issues](https://github.com/FunAudioLLM/FunCineForge/issues) or contact us for collaborative development.
153
+ For any questions, you can contact the [developer](mailto:jxliu@mail.ustc.edu.cn).
154
+
155
+ ### Disclaimer
156
+
157
+ This repository contains research artifacts:
158
+
159
+ ⚠️ Currently not a commercial product of Tongyi Lab.
160
+
161
+ ⚠️ Released for academic research / cutting-edge exploration purposes
162
+
163
+ ⚠️ CineDub Dataset samples are subject to specific license terms.
asd.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b020ff7104cad71e14a51c7cedfa614ded2de7befe5c73f88c50792a2933783c
3
+ size 63208524
config.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {}
face_recog_ir101.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1695521d026730358b7304a35542a86dad2aa4cad4f8bc25043975f4b6f679fb
3
+ size 260698833
fqa.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d0e02b72f987989b5fe0447745521e16067b10777b66a1fb89362fb1ca08183
3
+ size 406114
fun_2d.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:619a31681264d3f7f7fc7a16a42cbbe8b23f31a256f75a366e5a1bcd59b33543
3
+ size 89843225
fun_2d.zip ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd938726adb1f15f361263cce2db9cb820c42585fa8796ec72ce19107f369a46
3
+ size 96316515
funcineforge_zh_en/camplus.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a6ac6a63997761ae2997373e2ee1c47040854b4b759ea41ec48e4e42df0f4d73
3
+ size 28303423
funcineforge_zh_en/flow/config.yaml ADDED
@@ -0,0 +1,105 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ model: CosyVoiceFlowMatching
2
+ model_conf:
3
+ model_dtype: fp32
4
+ codebook_size: 6561
5
+ model_size: 1024
6
+ xvec_size: 192
7
+ feat_token_ratio: 2
8
+ mel_norm_type: null
9
+ lookahead_length: 3
10
+ training_cfg_rate: 0.2
11
+ inference_cfg_rate: 0.7
12
+ only_mask_loss: true
13
+ dit_conf:
14
+ dim: 1024
15
+ depth: 22
16
+ heads: 16
17
+ dim_head: 64
18
+ ff_mult: 2
19
+ mel_dim: 80
20
+ mu_dim: 80
21
+ spk_dim: 80
22
+ causal_mask_type:
23
+ - prob_min: 0
24
+ prob_max: 0.25
25
+ block_size: -1
26
+ ratio: 2
27
+ - prob_min: 0.25
28
+ prob_max: 0.5
29
+ block_size: 1
30
+ ratio: 2
31
+ - prob_min: 0.5
32
+ prob_max: 0.75
33
+ block_size: 15
34
+ ratio: 2
35
+ - prob_min: 0.75
36
+ prob_max: 1.0
37
+ block_size: 30
38
+ ratio: 2
39
+ mel_feat_conf:
40
+ n_fft: 1920
41
+ hop_length: 480
42
+ win_length: 1920
43
+ sampling_rate: 24000
44
+ n_mel_channels: 80
45
+ mel_fmin: 0
46
+ mel_fmax: 8000
47
+ center: false
48
+ feat_type: power_log
49
+ prompt_conf:
50
+ prompt_type: prefix
51
+ prompt_width_ratio_range:
52
+ - 0.7
53
+ - 1.0
54
+ frontend: WhisperFrontend
55
+ frontend_conf:
56
+ fs: 24000
57
+ n_mels: 80
58
+ do_pad_trim: false
59
+ filters_path:
60
+ train_conf:
61
+ use_lora: false
62
+ accum_grad: 1
63
+ grad_clip: 1
64
+ max_epoch: 150
65
+ keep_nbest_models: 150000
66
+ log_interval: 50
67
+ effective_save_name_excludes:
68
+ - none
69
+ resume: true
70
+ validate_interval: 10000
71
+ save_checkpoint_interval: 10000
72
+ avg_nbest_model: 100
73
+ use_bf16: false
74
+ use_deepspeed: true
75
+ save_init_model: false
76
+ loss_rescale_by_rank: false
77
+ deepspeed_config: decode_conf/ds_stage0_fp32.json
78
+ optim: adamw
79
+ optim_conf:
80
+ lr: 0.0001
81
+ scheduler: warmuplr
82
+ scheduler_conf:
83
+ warmup_steps: 10000
84
+ dataset: CosyVoiceFlowMetaDataset
85
+ dataset_conf:
86
+ wav_token_ratio: 960
87
+ load_meta_data_key: text,token,wav_path,spk_emb_path
88
+ set_invalid_xvec_zeros: true
89
+ index_ds: CosyVoice
90
+ data_split_num: 64
91
+ batch_sampler: BatchSampler
92
+ shuffle: true
93
+ sort_size: 512
94
+ batch_type: token
95
+ batch_size: 10000
96
+ batch_size_token_max: 12000
97
+ batch_size_sample_max: 100
98
+ max_token_length: 2250
99
+ max_text_length: null
100
+ batch_size_scale_threshold: 3000
101
+ num_workers: 6
102
+ retry: 100
103
+ enable_tf32: true
104
+ debug: false
105
+ device: cpu
funcineforge_zh_en/flow/ds-model.pt.best/mp_rank_00_model_states.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:54932cde43cb4beb54648b14ff701dd92eaa16423f463b594148dfaae6593a74
3
+ size 3987933779
funcineforge_zh_en/llm/config.yaml ADDED
@@ -0,0 +1,112 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ model: FunCineForgeLM
2
+ model_conf:
3
+ lsm_weight: 0.0
4
+ length_normalized_loss: true
5
+ codec_unit: 6761
6
+ timespk_unit: 1550
7
+ face_size: 512
8
+ llm: Qwen2-0.5B
9
+ llm_conf:
10
+ hub: hf
11
+ freeze: false
12
+ llm_dtype: fp32
13
+ init_param_path: ../tokenizer/Qwen2-0.5B-CosyVoice-BlankEN
14
+ use_lora: false
15
+ lora_conf:
16
+ task_type: CAUSAL_LM
17
+ r: 16
18
+ lora_alpha: 32
19
+ lora_dropout: 0.05
20
+ bias: none
21
+ target_modules:
22
+ - q_proj
23
+ - v_proj
24
+ train_conf:
25
+ use_lora: ${llm_conf.use_lora}
26
+ accum_grad: 1
27
+ grad_clip: 5
28
+ max_epoch: 200
29
+ log_interval: 100
30
+ effective_save_name_excludes:
31
+ - none
32
+ resume: true
33
+ validate_interval: 5000
34
+ save_checkpoint_interval: 5000
35
+ keep_nbest_models: 100000
36
+ avg_nbest_model: 5
37
+ use_bf16: false
38
+ save_init_model: false
39
+ loss_rescale_by_rank: false
40
+ use_deepspeed: true
41
+ deepspeed_config: decode_conf/ds_stage0_fp32.json
42
+ optim: adamw
43
+ optim_conf:
44
+ lr: 8.0e-05
45
+ scheduler: warmuplr
46
+ scheduler_conf:
47
+ warmup_steps: 2000
48
+ dataset: FunCineForgeDataset
49
+ dataset_conf:
50
+ use_emotion_clue: true
51
+ codebook_size: 6561
52
+ sos: 6561
53
+ eos: 6562
54
+ turn_of_speech: 6563
55
+ fill_token: 6564
56
+ ignore_id: -100
57
+ startofclue_token: 151646
58
+ endofclue_token: 151647
59
+ frame_shift: 25
60
+ timebook_size: 1500
61
+ pangbai: 1500
62
+ dubai: 1501
63
+ duihua: 1502
64
+ duoren: 1503
65
+ male: 1504
66
+ female: 1505
67
+ child: 1506
68
+ youth: 1507
69
+ adult: 1508
70
+ middle: 1509
71
+ elderly: 1510
72
+ speaker_id_start: 1511
73
+ index_ds: CosyVoice
74
+ dataloader: DataloaderMapStyle
75
+ load_meta_data_key: text,clue,token,face,dialogue
76
+ data_split_num: 1
77
+ batch_sampler: BatchSampler
78
+ shuffle: true
79
+ sort_size: 512
80
+ face_size: 512
81
+ batch_type: token
82
+ batch_size: 3000
83
+ batch_size_token_max: 20000
84
+ batch_size_sample_max: 100
85
+ max_token_length: 5000
86
+ max_text_length: 300
87
+ batch_size_scale_threshold: 3000
88
+ num_workers: 20
89
+ retry: 100
90
+ specaug: FunCineForgeSpecAug
91
+ specaug_conf:
92
+ apply_time_warp: false
93
+ apply_freq_mask: false
94
+ apply_time_mask: true
95
+ time_mask_width_ratio_range:
96
+ - 0
97
+ - 0.05
98
+ num_time_mask: 10
99
+ fill_value: -100
100
+ tokenizer: FunCineForgeTokenizer
101
+ tokenizer_conf:
102
+ init_param_path: ${llm_conf.init_param_path}
103
+ face_encoder: FaceRecIR101
104
+ face_encoder_conf:
105
+ init_param_path: ../speaker_diarization/pretrained_models/face_recog_ir101.onnx
106
+ enable_tf32: true
107
+ debug: false
108
+ train_data_set_list: /nfs/yanzhang.ljx/workspace/datasets/YingShi/clean/train.jsonl
109
+ valid_data_set_list: /nfs/yanzhang.ljx/workspace/datasets/YingShi/clean/test.jsonl
110
+ output_dir: /cpfs_fundata/yanzhang.ljx/workspace/exps/1m-8gpu/zh_en
111
+ init_param: /nfs/hengwu.zty/exps/4m-8gpu/CosyVoice_MixedAM_5b15_Qwen2_500M_phn_fp32_fsq6561_simple_sys_minmo_l12_merge_cosyvoice3d5_baiyinku_emilia_yodas2_0605/ds-model.pt.ep0.290000/mp_rank_00_model_states.pt
112
+ device: cpu
funcineforge_zh_en/llm/ds-model.pt.best/mp_rank_00_model_states.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2ef73ff7c19b85cadecf0ea134173100c44353b643322788778c3f687f1f5a20
3
+ size 6096417415
funcineforge_zh_en/vocoder/config.yaml ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ model: CausalHifiGan
2
+ model_conf:
3
+ CausalHiFTGenerator_conf:
4
+ in_channels: 80
5
+ base_channels: 512
6
+ nb_harmonics: 8
7
+ sampling_rate: 24000
8
+ nsf_alpha: 0.1
9
+ nsf_sigma: 0.003
10
+ nsf_voiced_threshold: 10
11
+ upsample_rates: [8, 5, 3]
12
+ upsample_kernel_sizes: [16, 11, 7]
13
+ istft_params:
14
+ n_fft: 16
15
+ hop_len: 4
16
+ resblock_kernel_sizes: [3, 7, 11]
17
+ resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
18
+ source_resblock_kernel_sizes: [7, 7, 11]
19
+ source_resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
20
+ lrelu_slope: 0.1
21
+ audio_limit: 0.99
22
+ CausalConvRNNF0Predictor_conf:
23
+ num_class: 1
24
+ in_channels: 80
25
+ cond_channels: 512
26
+ sample_rate: 24000
funcineforge_zh_en/vocoder/ds-model.pt.best/avg_5_removewn.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aac00e77b8bec73bdeebd2fa06b4bea531f396afa35f962d8dc1708c6e876d9f
3
+ size 83141596
funcineforge_zh_en/vocoder/hift_causal.hyper.yaml ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # set random seed, so that you may reproduce your result.
2
+ __set_seed1: !apply:random.seed [1986]
3
+ __set_seed2: !apply:numpy.random.seed [1986]
4
+ __set_seed3: !apply:torch.manual_seed [1986]
5
+ __set_seed4: !apply:torch.cuda.manual_seed_all [1986]
6
+
7
+ # fixed params
8
+ sample_rate: 24000
9
+ text_encoder_input_size: 512
10
+ llm_input_size: 1024
11
+ llm_output_size: 1024
12
+ spk_embed_dim: 192
13
+
14
+ # model params
15
+ # for all class/function included in this repo, we use !<name> or !<new> for intialization, so that user may find all corresponding class/function according to one single yaml.
16
+ hift: !new:cosyvoice.models.vocoder.hift_causal.CausalHiFTGenerator
17
+ in_channels: 80
18
+ base_channels: 512
19
+ nb_harmonics: 8
20
+ sampling_rate: !ref <sample_rate>
21
+ nsf_alpha: 0.1
22
+ nsf_sigma: 0.003
23
+ nsf_voiced_threshold: 10
24
+ upsample_rates: [8, 5, 3]
25
+ upsample_kernel_sizes: [16, 11, 7]
26
+ istft_params:
27
+ n_fft: 16
28
+ hop_len: 4
29
+ resblock_kernel_sizes: [3, 7, 11]
30
+ resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
31
+ source_resblock_kernel_sizes: [7, 7, 11]
32
+ source_resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
33
+ lrelu_slope: 0.1
34
+ audio_limit: 0.99
35
+ f0_predictor: !new:cosyvoice.models.vocoder.f0_predictor_causal.CausalConvRNNF0Predictor
36
+ num_class: 1
37
+ in_channels: 80
38
+ cond_channels: 512
speech_campplus/.mdl ADDED
Binary file (71 Bytes). View file
 
speech_campplus/.msc ADDED
Binary file (760 Bytes). View file
 
speech_campplus/.mv ADDED
@@ -0,0 +1 @@
 
 
1
+ Revision:v1.0.0,CreatedAt:1708583355
speech_campplus/campplus_cn_en_common.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:92f29b94e6948786a26778c9e302525d185bb08c8b9f5252ed98776902840199
3
+ size 28044640
speech_campplus/config.yaml ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # This is an example that demonstrates how to configure a model file.
2
+ # You can modify the configuration according to your own requirements.
3
+
4
+ # to print the register_table:
5
+ # from funasr.register import tables
6
+ # tables.print()
7
+
8
+ # network architecture
9
+ model: CAMPPlus
10
+ model_conf:
11
+ feat_dim: 80
12
+ embedding_size: 192
13
+ growth_rate: 32
14
+ bn_size: 4
15
+ init_channels: 128
16
+ config_str: 'batchnorm-relu'
17
+ memory_efficient: True
18
+ output_level: 'segment'
19
+
20
+ # frontend related
21
+ frontend: WavFrontend
22
+ frontend_conf:
23
+ fs: 16000
speech_campplus/configuration.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "framework": "pytorch",
3
+ "task": "speaker-verification",
4
+ "model_config": "config.yaml",
5
+ "model_file": "campplus_cn_en_common.pt",
6
+ "model": {
7
+ "type": "cam++-sv",
8
+ "model_config": {
9
+ "sample_rate": 16000,
10
+ "fbank_dim": 80,
11
+ "emb_size": 192
12
+ },
13
+ "pretrained_model": "campplus_cn_en_common.pt",
14
+ "yesOrno_thr": 0.33
15
+ },
16
+ "pipeline": {
17
+ "type": "speaker-verification"
18
+ },
19
+ "file_path_metas": {
20
+ "init_param":"campplus_cn_en_common.pt",
21
+ "config":"config.yaml"
22
+ }
23
+ }
speech_campplus/structure.png ADDED

Git LFS Details

  • SHA256: 1ff916275cbfe40e1e5584ef66f81b776ef992e9997d8658328394d023dba1b8
  • Pointer size: 131 Bytes
  • Size of remote file: 286 kB
speech_fsmn_vad/.mdl ADDED
Binary file (67 Bytes). View file
 
speech_fsmn_vad/.msc ADDED
Binary file (511 Bytes). View file
 
speech_fsmn_vad/.mv ADDED
@@ -0,0 +1 @@
 
 
1
+ Revision:v2.0.4,CreatedAt:1706001004
speech_fsmn_vad/am.mvn ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ <Nnet>
2
+ <Splice> 400 400
3
+ [ 0 ]
4
+ <AddShift> 400 400
5
+ <LearnRateCoef> 0 [ -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 ]
6
+ <Rescale> 400 400
7
+ <LearnRateCoef> 0 [ 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 ]
8
+ </Nnet>
speech_fsmn_vad/config.yaml ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ frontend: WavFrontendOnline
2
+ frontend_conf:
3
+ fs: 16000
4
+ window: hamming
5
+ n_mels: 80
6
+ frame_length: 25
7
+ frame_shift: 10
8
+ dither: 0.0
9
+ lfr_m: 5
10
+ lfr_n: 1
11
+
12
+ model: FsmnVADStreaming
13
+ model_conf:
14
+ sample_rate: 16000
15
+ detect_mode: 1
16
+ snr_mode: 0
17
+ max_end_silence_time: 800
18
+ max_start_silence_time: 3000
19
+ do_start_point_detection: True
20
+ do_end_point_detection: True
21
+ window_size_ms: 200
22
+ sil_to_speech_time_thres: 150
23
+ speech_to_sil_time_thres: 150
24
+ speech_2_noise_ratio: 1.0
25
+ do_extend: 1
26
+ lookback_time_start_point: 200
27
+ lookahead_time_end_point: 100
28
+ max_single_segment_time: 60000
29
+ snr_thres: -100.0
30
+ noise_frame_num_used_for_snr: 100
31
+ decibel_thres: -100.0
32
+ speech_noise_thres: 0.6
33
+ fe_prior_thres: 0.0001
34
+ silence_pdf_num: 1
35
+ sil_pdf_ids: [0]
36
+ speech_noise_thresh_low: -0.1
37
+ speech_noise_thresh_high: 0.3
38
+ output_frame_probs: False
39
+ frame_in_ms: 10
40
+ frame_length_ms: 25
41
+
42
+ encoder: FSMN
43
+ encoder_conf:
44
+ input_dim: 400
45
+ input_affine_dim: 140
46
+ fsmn_layers: 4
47
+ linear_dim: 250
48
+ proj_dim: 128
49
+ lorder: 20
50
+ rorder: 0
51
+ lstride: 1
52
+ rstride: 0
53
+ output_affine_dim: 140
54
+ output_dim: 248
55
+
56
+
speech_fsmn_vad/configuration.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "framework": "pytorch",
3
+ "task" : "voice-activity-detection",
4
+ "pipeline": {"type":"funasr-pipeline"},
5
+ "model": {"type" : "funasr"},
6
+ "file_path_metas": {
7
+ "init_param":"model.pt",
8
+ "config":"config.yaml",
9
+ "frontend_conf":{"cmvn_file": "am.mvn"}},
10
+ "model_name_in_hub": {
11
+ "ms":"iic/speech_fsmn_vad_zh-cn-16k-common-pytorch",
12
+ "hf":""}
13
+ }
speech_fsmn_vad/model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b3be75be477f0780277f3bae0fe489f48718f585f3a6e45d7dd1fbb1a4255fc5
3
+ size 1721366
speech_fsmn_vad/struct.png ADDED
speech_tokenizer_v3.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:23236a74175dbdda47afc66dbadd5bcb41303c467a57c261cb8539ad9db9208d
3
+ size 969451503
version-RFB-320.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:68bbbaa1023629ab4967c735c133ec2d440d94941cdcb4e0d9c9cd2ab0c83c0d
3
+ size 1231013