Upload 41 files
Browse files- .gitattributes +1 -0
- MelBandRoformer.ckpt +3 -0
- Qwen2-0.5B-CosyVoice-BlankEN/LICENSE +202 -0
- Qwen2-0.5B-CosyVoice-BlankEN/README.md +94 -0
- Qwen2-0.5B-CosyVoice-BlankEN/config.json +27 -0
- Qwen2-0.5B-CosyVoice-BlankEN/generation_config.json +14 -0
- Qwen2-0.5B-CosyVoice-BlankEN/merges.txt +0 -0
- Qwen2-0.5B-CosyVoice-BlankEN/model.safetensors +3 -0
- Qwen2-0.5B-CosyVoice-BlankEN/tokenizer_config.json +40 -0
- Qwen2-0.5B-CosyVoice-BlankEN/vocab.json +0 -0
- README.md +163 -0
- asd.onnx +3 -0
- config.json +1 -0
- face_recog_ir101.onnx +3 -0
- fqa.onnx +3 -0
- fun_2d.pth +3 -0
- fun_2d.zip +3 -0
- funcineforge_zh_en/camplus.onnx +3 -0
- funcineforge_zh_en/flow/config.yaml +105 -0
- funcineforge_zh_en/flow/ds-model.pt.best/mp_rank_00_model_states.pt +3 -0
- funcineforge_zh_en/llm/config.yaml +112 -0
- funcineforge_zh_en/llm/ds-model.pt.best/mp_rank_00_model_states.pt +3 -0
- funcineforge_zh_en/vocoder/config.yaml +26 -0
- funcineforge_zh_en/vocoder/ds-model.pt.best/avg_5_removewn.pt +3 -0
- funcineforge_zh_en/vocoder/hift_causal.hyper.yaml +38 -0
- speech_campplus/.mdl +0 -0
- speech_campplus/.msc +0 -0
- speech_campplus/.mv +1 -0
- speech_campplus/campplus_cn_en_common.pt +3 -0
- speech_campplus/config.yaml +23 -0
- speech_campplus/configuration.json +23 -0
- speech_campplus/structure.png +3 -0
- speech_fsmn_vad/.mdl +0 -0
- speech_fsmn_vad/.msc +0 -0
- speech_fsmn_vad/.mv +1 -0
- speech_fsmn_vad/am.mvn +8 -0
- speech_fsmn_vad/config.yaml +56 -0
- speech_fsmn_vad/configuration.json +13 -0
- speech_fsmn_vad/model.pt +3 -0
- speech_fsmn_vad/struct.png +0 -0
- speech_tokenizer_v3.onnx +3 -0
- version-RFB-320.onnx +3 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
speech_campplus/structure.png filter=lfs diff=lfs merge=lfs -text
|
MelBandRoformer.ckpt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:87201f4d31afb5bc79993230fc49446918425574db48c01c405e44f365c7559e
|
| 3 |
+
size 913106900
|
Qwen2-0.5B-CosyVoice-BlankEN/LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright 2024 Alibaba Cloud
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
Qwen2-0.5B-CosyVoice-BlankEN/README.md
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
language:
|
| 4 |
+
- en
|
| 5 |
+
pipeline_tag: text-generation
|
| 6 |
+
tags:
|
| 7 |
+
- chat
|
| 8 |
+
base_model: Qwen/Qwen2-0.5B
|
| 9 |
+
---
|
| 10 |
+
|
| 11 |
+
# Qwen2-0.5B-Instruct
|
| 12 |
+
|
| 13 |
+
## Introduction
|
| 14 |
+
|
| 15 |
+
Qwen2 is the new series of Qwen large language models. For Qwen2, we release a number of base language models and instruction-tuned language models ranging from 0.5 to 72 billion parameters, including a Mixture-of-Experts model. This repo contains the instruction-tuned 0.5B Qwen2 model.
|
| 16 |
+
|
| 17 |
+
Compared with the state-of-the-art opensource language models, including the previous released Qwen1.5, Qwen2 has generally surpassed most opensource models and demonstrated competitiveness against proprietary models across a series of benchmarks targeting for language understanding, language generation, multilingual capability, coding, mathematics, reasoning, etc.
|
| 18 |
+
|
| 19 |
+
For more details, please refer to our [blog](https://qwenlm.github.io/blog/qwen2/), [GitHub](https://github.com/QwenLM/Qwen2), and [Documentation](https://qwen.readthedocs.io/en/latest/).
|
| 20 |
+
<br>
|
| 21 |
+
|
| 22 |
+
## Model Details
|
| 23 |
+
Qwen2 is a language model series including decoder language models of different model sizes. For each size, we release the base language model and the aligned chat model. It is based on the Transformer architecture with SwiGLU activation, attention QKV bias, group query attention, etc. Additionally, we have an improved tokenizer adaptive to multiple natural languages and codes.
|
| 24 |
+
|
| 25 |
+
## Training details
|
| 26 |
+
We pretrained the models with a large amount of data, and we post-trained the models with both supervised finetuning and direct preference optimization.
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
## Requirements
|
| 30 |
+
The code of Qwen2 has been in the latest Hugging face transformers and we advise you to install `transformers>=4.37.0`, or you might encounter the following error:
|
| 31 |
+
```
|
| 32 |
+
KeyError: 'qwen2'
|
| 33 |
+
```
|
| 34 |
+
|
| 35 |
+
## Quickstart
|
| 36 |
+
|
| 37 |
+
Here provides a code snippet with `apply_chat_template` to show you how to load the tokenizer and model and how to generate contents.
|
| 38 |
+
|
| 39 |
+
```python
|
| 40 |
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 41 |
+
device = "cuda" # the device to load the model onto
|
| 42 |
+
|
| 43 |
+
model = AutoModelForCausalLM.from_pretrained(
|
| 44 |
+
"Qwen/Qwen2-0.5B-Instruct",
|
| 45 |
+
torch_dtype="auto",
|
| 46 |
+
device_map="auto"
|
| 47 |
+
)
|
| 48 |
+
tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen2-0.5B-Instruct")
|
| 49 |
+
|
| 50 |
+
prompt = "Give me a short introduction to large language model."
|
| 51 |
+
messages = [
|
| 52 |
+
{"role": "system", "content": "You are a helpful assistant."},
|
| 53 |
+
{"role": "user", "content": prompt}
|
| 54 |
+
]
|
| 55 |
+
text = tokenizer.apply_chat_template(
|
| 56 |
+
messages,
|
| 57 |
+
tokenize=False,
|
| 58 |
+
add_generation_prompt=True
|
| 59 |
+
)
|
| 60 |
+
model_inputs = tokenizer([text], return_tensors="pt").to(device)
|
| 61 |
+
|
| 62 |
+
generated_ids = model.generate(
|
| 63 |
+
model_inputs.input_ids,
|
| 64 |
+
max_new_tokens=512
|
| 65 |
+
)
|
| 66 |
+
generated_ids = [
|
| 67 |
+
output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids)
|
| 68 |
+
]
|
| 69 |
+
|
| 70 |
+
response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0]
|
| 71 |
+
```
|
| 72 |
+
|
| 73 |
+
## Evaluation
|
| 74 |
+
|
| 75 |
+
We briefly compare Qwen2-0.5B-Instruct with Qwen1.5-0.5B-Chat. The results are as follows:
|
| 76 |
+
|
| 77 |
+
| Datasets | Qwen1.5-0.5B-Chat | **Qwen2-0.5B-Instruct** | Qwen1.5-1.8B-Chat | **Qwen2-1.5B-Instruct** |
|
| 78 |
+
| :--- | :---: | :---: | :---: | :---: |
|
| 79 |
+
| MMLU | 35.0 | **37.9** | 43.7 | **52.4** |
|
| 80 |
+
| HumanEval | 9.1 | **17.1** | 25.0 | **37.8** |
|
| 81 |
+
| GSM8K | 11.3 | **40.1** | 35.3 | **61.6** |
|
| 82 |
+
| C-Eval | 37.2 | **45.2** | 55.3 | **63.8** |
|
| 83 |
+
| IFEval (Prompt Strict-Acc.) | 14.6 | **20.0** | 16.8 | **29.0** |
|
| 84 |
+
|
| 85 |
+
## Citation
|
| 86 |
+
|
| 87 |
+
If you find our work helpful, feel free to give us a cite.
|
| 88 |
+
|
| 89 |
+
```
|
| 90 |
+
@article{qwen2,
|
| 91 |
+
title={Qwen2 Technical Report},
|
| 92 |
+
year={2024}
|
| 93 |
+
}
|
| 94 |
+
```
|
Qwen2-0.5B-CosyVoice-BlankEN/config.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"Qwen2ForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_dropout": 0.0,
|
| 6 |
+
"bos_token_id": 151643,
|
| 7 |
+
"eos_token_id": 151645,
|
| 8 |
+
"hidden_act": "silu",
|
| 9 |
+
"hidden_size": 896,
|
| 10 |
+
"initializer_range": 0.02,
|
| 11 |
+
"intermediate_size": 4864,
|
| 12 |
+
"max_position_embeddings": 32768,
|
| 13 |
+
"max_window_layers": 24,
|
| 14 |
+
"model_type": "qwen2",
|
| 15 |
+
"num_attention_heads": 14,
|
| 16 |
+
"num_hidden_layers": 24,
|
| 17 |
+
"num_key_value_heads": 2,
|
| 18 |
+
"rms_norm_eps": 1e-06,
|
| 19 |
+
"rope_theta": 1000000.0,
|
| 20 |
+
"sliding_window": 32768,
|
| 21 |
+
"tie_word_embeddings": true,
|
| 22 |
+
"torch_dtype": "bfloat16",
|
| 23 |
+
"transformers_version": "4.40.1",
|
| 24 |
+
"use_cache": true,
|
| 25 |
+
"use_sliding_window": false,
|
| 26 |
+
"vocab_size": 151936
|
| 27 |
+
}
|
Qwen2-0.5B-CosyVoice-BlankEN/generation_config.json
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token_id": 151643,
|
| 3 |
+
"pad_token_id": 151643,
|
| 4 |
+
"do_sample": true,
|
| 5 |
+
"eos_token_id": [
|
| 6 |
+
151645,
|
| 7 |
+
151643
|
| 8 |
+
],
|
| 9 |
+
"repetition_penalty": 1.1,
|
| 10 |
+
"temperature": 0.7,
|
| 11 |
+
"top_p": 0.8,
|
| 12 |
+
"top_k": 20,
|
| 13 |
+
"transformers_version": "4.37.0"
|
| 14 |
+
}
|
Qwen2-0.5B-CosyVoice-BlankEN/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
Qwen2-0.5B-CosyVoice-BlankEN/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:130282af0dfa9fe5840737cc49a0d339d06075f83c5a315c3372c9a0740d0b96
|
| 3 |
+
size 988097824
|
Qwen2-0.5B-CosyVoice-BlankEN/tokenizer_config.json
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"151643": {
|
| 5 |
+
"content": "<|endoftext|>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": false,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"151644": {
|
| 13 |
+
"content": "<|im_start|>",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": false,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": true
|
| 19 |
+
},
|
| 20 |
+
"151645": {
|
| 21 |
+
"content": "<|im_end|>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": false,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": true
|
| 27 |
+
}
|
| 28 |
+
},
|
| 29 |
+
"additional_special_tokens": ["<|im_start|>", "<|im_end|>"],
|
| 30 |
+
"bos_token": null,
|
| 31 |
+
"chat_template": "{% for message in messages %}{% if loop.first and messages[0]['role'] != 'system' %}{{ '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}{% endif %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
|
| 32 |
+
"clean_up_tokenization_spaces": false,
|
| 33 |
+
"eos_token": "<|im_end|>",
|
| 34 |
+
"errors": "replace",
|
| 35 |
+
"model_max_length": 32768,
|
| 36 |
+
"pad_token": "<|endoftext|>",
|
| 37 |
+
"split_special_tokens": false,
|
| 38 |
+
"tokenizer_class": "Qwen2Tokenizer",
|
| 39 |
+
"unk_token": null
|
| 40 |
+
}
|
Qwen2-0.5B-CosyVoice-BlankEN/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
README.md
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
datasets:
|
| 4 |
+
- FunAudioLLM/CineDub-Example
|
| 5 |
+
language:
|
| 6 |
+
- zh
|
| 7 |
+
- en
|
| 8 |
+
tags:
|
| 9 |
+
- Dubbing-model
|
| 10 |
+
---
|
| 11 |
+
|
| 12 |
+
<p align="center">
|
| 13 |
+
<b>🎬 Fun-CineForge: A Unified Dataset Pipeline and Model for Zero-Shot Movie Dubbing in Diverse Cinematic Scenes</b>
|
| 14 |
+
</p>
|
| 15 |
+
|
| 16 |
+
<div align="center">
|
| 17 |
+
|
| 18 |
+

|
| 19 |
+
<a href=""><img src="https://img.shields.io/badge/OS-Linux-orange.svg"></a>
|
| 20 |
+
<a href=""><img src="https://img.shields.io/badge/Python->=3.8-aff.svg"></a>
|
| 21 |
+
<a href=""><img src="https://img.shields.io/badge/Pytorch->=2.1-blue"></a>
|
| 22 |
+
</div>
|
| 23 |
+
|
| 24 |
+
<div align="center">
|
| 25 |
+
<h4><a href="#Open-Source">Open Source</a>
|
| 26 |
+
|<a href="#Dataset-Pipeline">Dataset Pipeline</a>
|
| 27 |
+
|<a href="#Dubbing-Model">Dubbing Model</a>
|
| 28 |
+
|<a href="#Recent-Updates">Recent Updates</a>
|
| 29 |
+
|<a href="#Publication">Publication</a>
|
| 30 |
+
|<a href="#Comminicate">Comminicate</a>
|
| 31 |
+
</h4>
|
| 32 |
+
</div>
|
| 33 |
+
|
| 34 |
+
**Fun-CineForge** contains an end-to-end dataset pipeline for producing large-scale dubbing datasets and an MLLM-based dubbing model designed for diverse cinematic scenes.
|
| 35 |
+
Using this pipeline, we constructed the first large-scale Chinese television dubbing dataset CineDub-CN, which includes rich annotations and diverse scenes.
|
| 36 |
+
In monologue, narration, dialogue, and multi-speaker scenes, our dubbing model consistently outperforms state-of-the-art methods in terms of audio quality, lip-sync, timbre transition, and instruction following.
|
| 37 |
+
|
| 38 |
+
<a name="Open-Source"></a>
|
| 39 |
+
## Open Source 🎬
|
| 40 |
+
You can access [https://funcineforge.github.io/](https://funcineforge.github.io/) to get our CineDub-CN dataset samples and demo samples.
|
| 41 |
+
|
| 42 |
+
GitHub link: [https://github.com/FunAudioLLM/FunCineForge/](https://github.com/FunAudioLLM/FunCineForge/)
|
| 43 |
+
|
| 44 |
+
Modelscope link: [https://www.modelscope.cn/models/FunAudioLLM/Fun-CineForge/](https://www.modelscope.cn/models/FunAudioLLM/Fun-CineForge/)
|
| 45 |
+
|
| 46 |
+
CineDub Samples:
|
| 47 |
+
[huggingface](https://huggingface.co/datasets/FunAudioLLM/CineDub-Example/)
|
| 48 |
+
[modelscope](https://www.modelscope.cn/datasets/FunAudioLLM/CineDub-Example)
|
| 49 |
+
|
| 50 |
+
<a name="Dataset-Pipeline"></a>
|
| 51 |
+
## Dataset Pipeline 🔨
|
| 52 |
+
|
| 53 |
+
### Environmental Installation
|
| 54 |
+
|
| 55 |
+
Fun-CineForge dataset pipeline toolkit only relies on a Python environment to run.
|
| 56 |
+
```shell
|
| 57 |
+
# Conda
|
| 58 |
+
git clone git@github.com:FunAudioLLM/FunCineForge.git
|
| 59 |
+
conda create -n FunCineForge python=3.10 -y && conda activate FunCineForge
|
| 60 |
+
sudo apt-get install ffmpeg
|
| 61 |
+
# Initial settings
|
| 62 |
+
python setup.py
|
| 63 |
+
```
|
| 64 |
+
|
| 65 |
+
### Data collection
|
| 66 |
+
If you want to produce your own data,
|
| 67 |
+
we recommend that you refer to the following requirements to collect the corresponding movies or television series.
|
| 68 |
+
|
| 69 |
+
1. Video source: TV dramas or movies, non documentaries, with more monologues or dialogue scenes, clear and unobstructed faces (such as without masks and veils).
|
| 70 |
+
2. Speech Requirements: Standard pronunciation, clear articulation, prominent human voice. Avoid materials with strong dialects, excessive background noise, or strong colloquialism.
|
| 71 |
+
3. Image Requirements: High resolution, clear facial details, sufficient lighting, avoiding extremely dark or strong backlit scenes.
|
| 72 |
+
|
| 73 |
+
### How to use
|
| 74 |
+
|
| 75 |
+
- [1] Standardize video format and name; trim the beginning and end of long videos; extract the audio from the trimmed video. (default is to trim 10 seconds from both the beginning and end.)
|
| 76 |
+
```shell
|
| 77 |
+
python normalize_trim.py --root datasets/raw_zh --intro 10 --outro 10
|
| 78 |
+
```
|
| 79 |
+
|
| 80 |
+
- [2] [Speech Separation](./speech_separation/README.md). The audio is used to separate the vocals from the instrumental music.
|
| 81 |
+
```shell
|
| 82 |
+
cd speech_separation
|
| 83 |
+
python run.py --root datasets/clean/zh --gpus 0 1 2 3
|
| 84 |
+
```
|
| 85 |
+
|
| 86 |
+
- [3] [VideoClipper](./video_clip/README.md). For long videos, VideoClipper is used to obtain sentence-level subtitle files and clip the long video into segments based on timestamps. Now it supports bilingualism in both Chinese and English. Below is an example in Chinese. It is recommended to use gpu acceleration for English.
|
| 87 |
+
```shell
|
| 88 |
+
cd video_clip
|
| 89 |
+
bash run.sh --stage 1 --stop_stage 2 --input datasets/raw_zh --output datasets/clean/zh --lang zh --device cpu
|
| 90 |
+
```
|
| 91 |
+
|
| 92 |
+
- Video duration limit and check for cleanup. (Without --execute, only pre-deleted files will be printed. After checking, add --execute to confirm the deletion.)
|
| 93 |
+
```shell
|
| 94 |
+
python clean_video.py --root datasets/clean/zh
|
| 95 |
+
python clean_srt.py --root datasets/clean/zh --lang zh
|
| 96 |
+
```
|
| 97 |
+
|
| 98 |
+
- [4] [Speaker Diarization](./speaker_diarization/README.md). Multimodal active speaker recognition obtains RTTM files; identifies the speaker's facial frames, extracts frame-level speaker face and lip raw data.
|
| 99 |
+
```shell
|
| 100 |
+
cd speaker_diarization
|
| 101 |
+
bash run.sh --stage 1 --stop_stage 4 --hf_access_token hf_xxx --root datasets/clean/zh --gpus "0 1 2 3"
|
| 102 |
+
```
|
| 103 |
+
|
| 104 |
+
- [5] Multimodal CoT Correction. Based on general-purpose MLLMs, the system uses audio, ASR text, and RTTM files as input. It leverages Chain-of-Thought (CoT) reasoning to extract clues and corrects the results of the specialized models. It also annotates character age, gender, and vocal timbre. Experimental results show that this strategy reduces the CER from 4.53% to 0.94% and the speaker diarization error rate from 8.38% to 1.20%, achieving quality comparable to or even better than manual transcription. Adding the --resume enables breakpoint COT inference to prevent wasted resources from repeated COT inferences. Now supports both Chinese and English.
|
| 105 |
+
```shell
|
| 106 |
+
python cot.py --root_dir datasets/clean/zh --lang zh --provider google --model gemini-3-pro-preview --api_key xxx --resume
|
| 107 |
+
python cot.py --root_dir datasets/clean/en --lang en --provider google --model gemini-3-pro-preview --api_key xxx --resume
|
| 108 |
+
python build_datasets.py --root_zh datasets/clean/zh --root_en datasets/clean/en --out_dir datasets/clean --save
|
| 109 |
+
```
|
| 110 |
+
|
| 111 |
+
- (Reference) Extract speech tokens based on the CosyVoice3 tokenizer for llm training.
|
| 112 |
+
```shell
|
| 113 |
+
python speech_tokenizer.py --root datasets/clean/zh
|
| 114 |
+
```
|
| 115 |
+
|
| 116 |
+
<a name="Dubbing-Model"></a>
|
| 117 |
+
## Dubbing Model ⚙️
|
| 118 |
+
We've open-sourced the inference code and the **infer.sh** script, and provided some test cases in the data folder for your experience. Inference requires a consumer-grade GPU. Run the following command:
|
| 119 |
+
|
| 120 |
+
```shell
|
| 121 |
+
cd exps
|
| 122 |
+
bash infer.sh
|
| 123 |
+
```
|
| 124 |
+
|
| 125 |
+
The API for multi-speaker dubbing from raw videos and SRT scripts is under development ...
|
| 126 |
+
|
| 127 |
+
<a name="Recent-Updates"></a>
|
| 128 |
+
## Recent Updates 🚀
|
| 129 |
+
- 2025/12/18: Fun-CineForge dataset pipeline toolkit is online! 🔥
|
| 130 |
+
- 2026/01/19: Chinese demo samples and CineDub-CN dataset samples released. 🔥
|
| 131 |
+
- 2026/01/25: Fix some environmental and operational issues.
|
| 132 |
+
- 2026/02/09: Optimized the data pipeline and added support for English videos.
|
| 133 |
+
- 2026/03/05: English demo samples and CineDub-EN dataset samples released. 🔥
|
| 134 |
+
- 2026/03/16: Open source inference code and checkpoints. 🔥
|
| 135 |
+
|
| 136 |
+
<a name="Publication"></a>
|
| 137 |
+
## Publication 📚
|
| 138 |
+
If you use our dataset or code, please cite the following paper:
|
| 139 |
+
<pre>
|
| 140 |
+
@misc{liu2026funcineforgeunifieddatasettoolkit,
|
| 141 |
+
title={FunCineForge: A Unified Dataset Toolkit and Model for Zero-Shot Movie Dubbing in Diverse Cinematic Scenes},
|
| 142 |
+
author={Jiaxuan Liu and Yang Xiang and Han Zhao and Xiangang Li and Zhenhua Ling},
|
| 143 |
+
year={2026},
|
| 144 |
+
eprint={2601.14777},
|
| 145 |
+
archivePrefix={arXiv},
|
| 146 |
+
primaryClass={cs.CV},
|
| 147 |
+
}
|
| 148 |
+
</pre>
|
| 149 |
+
|
| 150 |
+
<a name="Comminicate"></a>
|
| 151 |
+
## Comminicate 🍟
|
| 152 |
+
We welcome you to participate in discussions on Fun-CineForge [GitHub Issues](https://github.com/FunAudioLLM/FunCineForge/issues) or contact us for collaborative development.
|
| 153 |
+
For any questions, you can contact the [developer](mailto:jxliu@mail.ustc.edu.cn).
|
| 154 |
+
|
| 155 |
+
### Disclaimer
|
| 156 |
+
|
| 157 |
+
This repository contains research artifacts:
|
| 158 |
+
|
| 159 |
+
⚠️ Currently not a commercial product of Tongyi Lab.
|
| 160 |
+
|
| 161 |
+
⚠️ Released for academic research / cutting-edge exploration purposes
|
| 162 |
+
|
| 163 |
+
⚠️ CineDub Dataset samples are subject to specific license terms.
|
asd.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b020ff7104cad71e14a51c7cedfa614ded2de7befe5c73f88c50792a2933783c
|
| 3 |
+
size 63208524
|
config.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{}
|
face_recog_ir101.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1695521d026730358b7304a35542a86dad2aa4cad4f8bc25043975f4b6f679fb
|
| 3 |
+
size 260698833
|
fqa.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4d0e02b72f987989b5fe0447745521e16067b10777b66a1fb89362fb1ca08183
|
| 3 |
+
size 406114
|
fun_2d.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:619a31681264d3f7f7fc7a16a42cbbe8b23f31a256f75a366e5a1bcd59b33543
|
| 3 |
+
size 89843225
|
fun_2d.zip
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cd938726adb1f15f361263cce2db9cb820c42585fa8796ec72ce19107f369a46
|
| 3 |
+
size 96316515
|
funcineforge_zh_en/camplus.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a6ac6a63997761ae2997373e2ee1c47040854b4b759ea41ec48e4e42df0f4d73
|
| 3 |
+
size 28303423
|
funcineforge_zh_en/flow/config.yaml
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
model: CosyVoiceFlowMatching
|
| 2 |
+
model_conf:
|
| 3 |
+
model_dtype: fp32
|
| 4 |
+
codebook_size: 6561
|
| 5 |
+
model_size: 1024
|
| 6 |
+
xvec_size: 192
|
| 7 |
+
feat_token_ratio: 2
|
| 8 |
+
mel_norm_type: null
|
| 9 |
+
lookahead_length: 3
|
| 10 |
+
training_cfg_rate: 0.2
|
| 11 |
+
inference_cfg_rate: 0.7
|
| 12 |
+
only_mask_loss: true
|
| 13 |
+
dit_conf:
|
| 14 |
+
dim: 1024
|
| 15 |
+
depth: 22
|
| 16 |
+
heads: 16
|
| 17 |
+
dim_head: 64
|
| 18 |
+
ff_mult: 2
|
| 19 |
+
mel_dim: 80
|
| 20 |
+
mu_dim: 80
|
| 21 |
+
spk_dim: 80
|
| 22 |
+
causal_mask_type:
|
| 23 |
+
- prob_min: 0
|
| 24 |
+
prob_max: 0.25
|
| 25 |
+
block_size: -1
|
| 26 |
+
ratio: 2
|
| 27 |
+
- prob_min: 0.25
|
| 28 |
+
prob_max: 0.5
|
| 29 |
+
block_size: 1
|
| 30 |
+
ratio: 2
|
| 31 |
+
- prob_min: 0.5
|
| 32 |
+
prob_max: 0.75
|
| 33 |
+
block_size: 15
|
| 34 |
+
ratio: 2
|
| 35 |
+
- prob_min: 0.75
|
| 36 |
+
prob_max: 1.0
|
| 37 |
+
block_size: 30
|
| 38 |
+
ratio: 2
|
| 39 |
+
mel_feat_conf:
|
| 40 |
+
n_fft: 1920
|
| 41 |
+
hop_length: 480
|
| 42 |
+
win_length: 1920
|
| 43 |
+
sampling_rate: 24000
|
| 44 |
+
n_mel_channels: 80
|
| 45 |
+
mel_fmin: 0
|
| 46 |
+
mel_fmax: 8000
|
| 47 |
+
center: false
|
| 48 |
+
feat_type: power_log
|
| 49 |
+
prompt_conf:
|
| 50 |
+
prompt_type: prefix
|
| 51 |
+
prompt_width_ratio_range:
|
| 52 |
+
- 0.7
|
| 53 |
+
- 1.0
|
| 54 |
+
frontend: WhisperFrontend
|
| 55 |
+
frontend_conf:
|
| 56 |
+
fs: 24000
|
| 57 |
+
n_mels: 80
|
| 58 |
+
do_pad_trim: false
|
| 59 |
+
filters_path:
|
| 60 |
+
train_conf:
|
| 61 |
+
use_lora: false
|
| 62 |
+
accum_grad: 1
|
| 63 |
+
grad_clip: 1
|
| 64 |
+
max_epoch: 150
|
| 65 |
+
keep_nbest_models: 150000
|
| 66 |
+
log_interval: 50
|
| 67 |
+
effective_save_name_excludes:
|
| 68 |
+
- none
|
| 69 |
+
resume: true
|
| 70 |
+
validate_interval: 10000
|
| 71 |
+
save_checkpoint_interval: 10000
|
| 72 |
+
avg_nbest_model: 100
|
| 73 |
+
use_bf16: false
|
| 74 |
+
use_deepspeed: true
|
| 75 |
+
save_init_model: false
|
| 76 |
+
loss_rescale_by_rank: false
|
| 77 |
+
deepspeed_config: decode_conf/ds_stage0_fp32.json
|
| 78 |
+
optim: adamw
|
| 79 |
+
optim_conf:
|
| 80 |
+
lr: 0.0001
|
| 81 |
+
scheduler: warmuplr
|
| 82 |
+
scheduler_conf:
|
| 83 |
+
warmup_steps: 10000
|
| 84 |
+
dataset: CosyVoiceFlowMetaDataset
|
| 85 |
+
dataset_conf:
|
| 86 |
+
wav_token_ratio: 960
|
| 87 |
+
load_meta_data_key: text,token,wav_path,spk_emb_path
|
| 88 |
+
set_invalid_xvec_zeros: true
|
| 89 |
+
index_ds: CosyVoice
|
| 90 |
+
data_split_num: 64
|
| 91 |
+
batch_sampler: BatchSampler
|
| 92 |
+
shuffle: true
|
| 93 |
+
sort_size: 512
|
| 94 |
+
batch_type: token
|
| 95 |
+
batch_size: 10000
|
| 96 |
+
batch_size_token_max: 12000
|
| 97 |
+
batch_size_sample_max: 100
|
| 98 |
+
max_token_length: 2250
|
| 99 |
+
max_text_length: null
|
| 100 |
+
batch_size_scale_threshold: 3000
|
| 101 |
+
num_workers: 6
|
| 102 |
+
retry: 100
|
| 103 |
+
enable_tf32: true
|
| 104 |
+
debug: false
|
| 105 |
+
device: cpu
|
funcineforge_zh_en/flow/ds-model.pt.best/mp_rank_00_model_states.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:54932cde43cb4beb54648b14ff701dd92eaa16423f463b594148dfaae6593a74
|
| 3 |
+
size 3987933779
|
funcineforge_zh_en/llm/config.yaml
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
model: FunCineForgeLM
|
| 2 |
+
model_conf:
|
| 3 |
+
lsm_weight: 0.0
|
| 4 |
+
length_normalized_loss: true
|
| 5 |
+
codec_unit: 6761
|
| 6 |
+
timespk_unit: 1550
|
| 7 |
+
face_size: 512
|
| 8 |
+
llm: Qwen2-0.5B
|
| 9 |
+
llm_conf:
|
| 10 |
+
hub: hf
|
| 11 |
+
freeze: false
|
| 12 |
+
llm_dtype: fp32
|
| 13 |
+
init_param_path: ../tokenizer/Qwen2-0.5B-CosyVoice-BlankEN
|
| 14 |
+
use_lora: false
|
| 15 |
+
lora_conf:
|
| 16 |
+
task_type: CAUSAL_LM
|
| 17 |
+
r: 16
|
| 18 |
+
lora_alpha: 32
|
| 19 |
+
lora_dropout: 0.05
|
| 20 |
+
bias: none
|
| 21 |
+
target_modules:
|
| 22 |
+
- q_proj
|
| 23 |
+
- v_proj
|
| 24 |
+
train_conf:
|
| 25 |
+
use_lora: ${llm_conf.use_lora}
|
| 26 |
+
accum_grad: 1
|
| 27 |
+
grad_clip: 5
|
| 28 |
+
max_epoch: 200
|
| 29 |
+
log_interval: 100
|
| 30 |
+
effective_save_name_excludes:
|
| 31 |
+
- none
|
| 32 |
+
resume: true
|
| 33 |
+
validate_interval: 5000
|
| 34 |
+
save_checkpoint_interval: 5000
|
| 35 |
+
keep_nbest_models: 100000
|
| 36 |
+
avg_nbest_model: 5
|
| 37 |
+
use_bf16: false
|
| 38 |
+
save_init_model: false
|
| 39 |
+
loss_rescale_by_rank: false
|
| 40 |
+
use_deepspeed: true
|
| 41 |
+
deepspeed_config: decode_conf/ds_stage0_fp32.json
|
| 42 |
+
optim: adamw
|
| 43 |
+
optim_conf:
|
| 44 |
+
lr: 8.0e-05
|
| 45 |
+
scheduler: warmuplr
|
| 46 |
+
scheduler_conf:
|
| 47 |
+
warmup_steps: 2000
|
| 48 |
+
dataset: FunCineForgeDataset
|
| 49 |
+
dataset_conf:
|
| 50 |
+
use_emotion_clue: true
|
| 51 |
+
codebook_size: 6561
|
| 52 |
+
sos: 6561
|
| 53 |
+
eos: 6562
|
| 54 |
+
turn_of_speech: 6563
|
| 55 |
+
fill_token: 6564
|
| 56 |
+
ignore_id: -100
|
| 57 |
+
startofclue_token: 151646
|
| 58 |
+
endofclue_token: 151647
|
| 59 |
+
frame_shift: 25
|
| 60 |
+
timebook_size: 1500
|
| 61 |
+
pangbai: 1500
|
| 62 |
+
dubai: 1501
|
| 63 |
+
duihua: 1502
|
| 64 |
+
duoren: 1503
|
| 65 |
+
male: 1504
|
| 66 |
+
female: 1505
|
| 67 |
+
child: 1506
|
| 68 |
+
youth: 1507
|
| 69 |
+
adult: 1508
|
| 70 |
+
middle: 1509
|
| 71 |
+
elderly: 1510
|
| 72 |
+
speaker_id_start: 1511
|
| 73 |
+
index_ds: CosyVoice
|
| 74 |
+
dataloader: DataloaderMapStyle
|
| 75 |
+
load_meta_data_key: text,clue,token,face,dialogue
|
| 76 |
+
data_split_num: 1
|
| 77 |
+
batch_sampler: BatchSampler
|
| 78 |
+
shuffle: true
|
| 79 |
+
sort_size: 512
|
| 80 |
+
face_size: 512
|
| 81 |
+
batch_type: token
|
| 82 |
+
batch_size: 3000
|
| 83 |
+
batch_size_token_max: 20000
|
| 84 |
+
batch_size_sample_max: 100
|
| 85 |
+
max_token_length: 5000
|
| 86 |
+
max_text_length: 300
|
| 87 |
+
batch_size_scale_threshold: 3000
|
| 88 |
+
num_workers: 20
|
| 89 |
+
retry: 100
|
| 90 |
+
specaug: FunCineForgeSpecAug
|
| 91 |
+
specaug_conf:
|
| 92 |
+
apply_time_warp: false
|
| 93 |
+
apply_freq_mask: false
|
| 94 |
+
apply_time_mask: true
|
| 95 |
+
time_mask_width_ratio_range:
|
| 96 |
+
- 0
|
| 97 |
+
- 0.05
|
| 98 |
+
num_time_mask: 10
|
| 99 |
+
fill_value: -100
|
| 100 |
+
tokenizer: FunCineForgeTokenizer
|
| 101 |
+
tokenizer_conf:
|
| 102 |
+
init_param_path: ${llm_conf.init_param_path}
|
| 103 |
+
face_encoder: FaceRecIR101
|
| 104 |
+
face_encoder_conf:
|
| 105 |
+
init_param_path: ../speaker_diarization/pretrained_models/face_recog_ir101.onnx
|
| 106 |
+
enable_tf32: true
|
| 107 |
+
debug: false
|
| 108 |
+
train_data_set_list: /nfs/yanzhang.ljx/workspace/datasets/YingShi/clean/train.jsonl
|
| 109 |
+
valid_data_set_list: /nfs/yanzhang.ljx/workspace/datasets/YingShi/clean/test.jsonl
|
| 110 |
+
output_dir: /cpfs_fundata/yanzhang.ljx/workspace/exps/1m-8gpu/zh_en
|
| 111 |
+
init_param: /nfs/hengwu.zty/exps/4m-8gpu/CosyVoice_MixedAM_5b15_Qwen2_500M_phn_fp32_fsq6561_simple_sys_minmo_l12_merge_cosyvoice3d5_baiyinku_emilia_yodas2_0605/ds-model.pt.ep0.290000/mp_rank_00_model_states.pt
|
| 112 |
+
device: cpu
|
funcineforge_zh_en/llm/ds-model.pt.best/mp_rank_00_model_states.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2ef73ff7c19b85cadecf0ea134173100c44353b643322788778c3f687f1f5a20
|
| 3 |
+
size 6096417415
|
funcineforge_zh_en/vocoder/config.yaml
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
model: CausalHifiGan
|
| 2 |
+
model_conf:
|
| 3 |
+
CausalHiFTGenerator_conf:
|
| 4 |
+
in_channels: 80
|
| 5 |
+
base_channels: 512
|
| 6 |
+
nb_harmonics: 8
|
| 7 |
+
sampling_rate: 24000
|
| 8 |
+
nsf_alpha: 0.1
|
| 9 |
+
nsf_sigma: 0.003
|
| 10 |
+
nsf_voiced_threshold: 10
|
| 11 |
+
upsample_rates: [8, 5, 3]
|
| 12 |
+
upsample_kernel_sizes: [16, 11, 7]
|
| 13 |
+
istft_params:
|
| 14 |
+
n_fft: 16
|
| 15 |
+
hop_len: 4
|
| 16 |
+
resblock_kernel_sizes: [3, 7, 11]
|
| 17 |
+
resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
|
| 18 |
+
source_resblock_kernel_sizes: [7, 7, 11]
|
| 19 |
+
source_resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
|
| 20 |
+
lrelu_slope: 0.1
|
| 21 |
+
audio_limit: 0.99
|
| 22 |
+
CausalConvRNNF0Predictor_conf:
|
| 23 |
+
num_class: 1
|
| 24 |
+
in_channels: 80
|
| 25 |
+
cond_channels: 512
|
| 26 |
+
sample_rate: 24000
|
funcineforge_zh_en/vocoder/ds-model.pt.best/avg_5_removewn.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aac00e77b8bec73bdeebd2fa06b4bea531f396afa35f962d8dc1708c6e876d9f
|
| 3 |
+
size 83141596
|
funcineforge_zh_en/vocoder/hift_causal.hyper.yaml
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# set random seed, so that you may reproduce your result.
|
| 2 |
+
__set_seed1: !apply:random.seed [1986]
|
| 3 |
+
__set_seed2: !apply:numpy.random.seed [1986]
|
| 4 |
+
__set_seed3: !apply:torch.manual_seed [1986]
|
| 5 |
+
__set_seed4: !apply:torch.cuda.manual_seed_all [1986]
|
| 6 |
+
|
| 7 |
+
# fixed params
|
| 8 |
+
sample_rate: 24000
|
| 9 |
+
text_encoder_input_size: 512
|
| 10 |
+
llm_input_size: 1024
|
| 11 |
+
llm_output_size: 1024
|
| 12 |
+
spk_embed_dim: 192
|
| 13 |
+
|
| 14 |
+
# model params
|
| 15 |
+
# for all class/function included in this repo, we use !<name> or !<new> for intialization, so that user may find all corresponding class/function according to one single yaml.
|
| 16 |
+
hift: !new:cosyvoice.models.vocoder.hift_causal.CausalHiFTGenerator
|
| 17 |
+
in_channels: 80
|
| 18 |
+
base_channels: 512
|
| 19 |
+
nb_harmonics: 8
|
| 20 |
+
sampling_rate: !ref <sample_rate>
|
| 21 |
+
nsf_alpha: 0.1
|
| 22 |
+
nsf_sigma: 0.003
|
| 23 |
+
nsf_voiced_threshold: 10
|
| 24 |
+
upsample_rates: [8, 5, 3]
|
| 25 |
+
upsample_kernel_sizes: [16, 11, 7]
|
| 26 |
+
istft_params:
|
| 27 |
+
n_fft: 16
|
| 28 |
+
hop_len: 4
|
| 29 |
+
resblock_kernel_sizes: [3, 7, 11]
|
| 30 |
+
resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
|
| 31 |
+
source_resblock_kernel_sizes: [7, 7, 11]
|
| 32 |
+
source_resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
|
| 33 |
+
lrelu_slope: 0.1
|
| 34 |
+
audio_limit: 0.99
|
| 35 |
+
f0_predictor: !new:cosyvoice.models.vocoder.f0_predictor_causal.CausalConvRNNF0Predictor
|
| 36 |
+
num_class: 1
|
| 37 |
+
in_channels: 80
|
| 38 |
+
cond_channels: 512
|
speech_campplus/.mdl
ADDED
|
Binary file (71 Bytes). View file
|
|
|
speech_campplus/.msc
ADDED
|
Binary file (760 Bytes). View file
|
|
|
speech_campplus/.mv
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
Revision:v1.0.0,CreatedAt:1708583355
|
speech_campplus/campplus_cn_en_common.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:92f29b94e6948786a26778c9e302525d185bb08c8b9f5252ed98776902840199
|
| 3 |
+
size 28044640
|
speech_campplus/config.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# This is an example that demonstrates how to configure a model file.
|
| 2 |
+
# You can modify the configuration according to your own requirements.
|
| 3 |
+
|
| 4 |
+
# to print the register_table:
|
| 5 |
+
# from funasr.register import tables
|
| 6 |
+
# tables.print()
|
| 7 |
+
|
| 8 |
+
# network architecture
|
| 9 |
+
model: CAMPPlus
|
| 10 |
+
model_conf:
|
| 11 |
+
feat_dim: 80
|
| 12 |
+
embedding_size: 192
|
| 13 |
+
growth_rate: 32
|
| 14 |
+
bn_size: 4
|
| 15 |
+
init_channels: 128
|
| 16 |
+
config_str: 'batchnorm-relu'
|
| 17 |
+
memory_efficient: True
|
| 18 |
+
output_level: 'segment'
|
| 19 |
+
|
| 20 |
+
# frontend related
|
| 21 |
+
frontend: WavFrontend
|
| 22 |
+
frontend_conf:
|
| 23 |
+
fs: 16000
|
speech_campplus/configuration.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"framework": "pytorch",
|
| 3 |
+
"task": "speaker-verification",
|
| 4 |
+
"model_config": "config.yaml",
|
| 5 |
+
"model_file": "campplus_cn_en_common.pt",
|
| 6 |
+
"model": {
|
| 7 |
+
"type": "cam++-sv",
|
| 8 |
+
"model_config": {
|
| 9 |
+
"sample_rate": 16000,
|
| 10 |
+
"fbank_dim": 80,
|
| 11 |
+
"emb_size": 192
|
| 12 |
+
},
|
| 13 |
+
"pretrained_model": "campplus_cn_en_common.pt",
|
| 14 |
+
"yesOrno_thr": 0.33
|
| 15 |
+
},
|
| 16 |
+
"pipeline": {
|
| 17 |
+
"type": "speaker-verification"
|
| 18 |
+
},
|
| 19 |
+
"file_path_metas": {
|
| 20 |
+
"init_param":"campplus_cn_en_common.pt",
|
| 21 |
+
"config":"config.yaml"
|
| 22 |
+
}
|
| 23 |
+
}
|
speech_campplus/structure.png
ADDED
|
Git LFS Details
|
speech_fsmn_vad/.mdl
ADDED
|
Binary file (67 Bytes). View file
|
|
|
speech_fsmn_vad/.msc
ADDED
|
Binary file (511 Bytes). View file
|
|
|
speech_fsmn_vad/.mv
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
Revision:v2.0.4,CreatedAt:1706001004
|
speech_fsmn_vad/am.mvn
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<Nnet>
|
| 2 |
+
<Splice> 400 400
|
| 3 |
+
[ 0 ]
|
| 4 |
+
<AddShift> 400 400
|
| 5 |
+
<LearnRateCoef> 0 [ -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 -8.311879 -8.600912 -9.615928 -10.43595 -11.21292 -11.88333 -12.36243 -12.63706 -12.8818 -12.83066 -12.89103 -12.95666 -13.19763 -13.40598 -13.49113 -13.5546 -13.55639 -13.51915 -13.68284 -13.53289 -13.42107 -13.65519 -13.50713 -13.75251 -13.76715 -13.87408 -13.73109 -13.70412 -13.56073 -13.53488 -13.54895 -13.56228 -13.59408 -13.62047 -13.64198 -13.66109 -13.62669 -13.58297 -13.57387 -13.4739 -13.53063 -13.48348 -13.61047 -13.64716 -13.71546 -13.79184 -13.90614 -14.03098 -14.18205 -14.35881 -14.48419 -14.60172 -14.70591 -14.83362 -14.92122 -15.00622 -15.05122 -15.03119 -14.99028 -14.92302 -14.86927 -14.82691 -14.7972 -14.76909 -14.71356 -14.61277 -14.51696 -14.42252 -14.36405 -14.30451 -14.23161 -14.19851 -14.16633 -14.15649 -14.10504 -13.99518 -13.79562 -13.3996 -12.7767 -11.71208 ]
|
| 6 |
+
<Rescale> 400 400
|
| 7 |
+
<LearnRateCoef> 0 [ 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 0.155775 0.154484 0.1527379 0.1518718 0.1506028 0.1489256 0.147067 0.1447061 0.1436307 0.1443568 0.1451849 0.1455157 0.1452821 0.1445717 0.1439195 0.1435867 0.1436018 0.1438781 0.1442086 0.1448844 0.1454756 0.145663 0.146268 0.1467386 0.1472724 0.147664 0.1480913 0.1483739 0.1488841 0.1493636 0.1497088 0.1500379 0.1502916 0.1505389 0.1506787 0.1507102 0.1505992 0.1505445 0.1505938 0.1508133 0.1509569 0.1512396 0.1514625 0.1516195 0.1516156 0.1515561 0.1514966 0.1513976 0.1512612 0.151076 0.1510596 0.1510431 0.151077 0.1511168 0.1511917 0.151023 0.1508045 0.1505885 0.1503493 0.1502373 0.1501726 0.1500762 0.1500065 0.1499782 0.150057 0.1502658 0.150469 0.1505335 0.1505505 0.1505328 0.1504275 0.1502438 0.1499674 0.1497118 0.1494661 0.1493102 0.1493681 0.1495501 0.1499738 0.1509654 ]
|
| 8 |
+
</Nnet>
|
speech_fsmn_vad/config.yaml
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
frontend: WavFrontendOnline
|
| 2 |
+
frontend_conf:
|
| 3 |
+
fs: 16000
|
| 4 |
+
window: hamming
|
| 5 |
+
n_mels: 80
|
| 6 |
+
frame_length: 25
|
| 7 |
+
frame_shift: 10
|
| 8 |
+
dither: 0.0
|
| 9 |
+
lfr_m: 5
|
| 10 |
+
lfr_n: 1
|
| 11 |
+
|
| 12 |
+
model: FsmnVADStreaming
|
| 13 |
+
model_conf:
|
| 14 |
+
sample_rate: 16000
|
| 15 |
+
detect_mode: 1
|
| 16 |
+
snr_mode: 0
|
| 17 |
+
max_end_silence_time: 800
|
| 18 |
+
max_start_silence_time: 3000
|
| 19 |
+
do_start_point_detection: True
|
| 20 |
+
do_end_point_detection: True
|
| 21 |
+
window_size_ms: 200
|
| 22 |
+
sil_to_speech_time_thres: 150
|
| 23 |
+
speech_to_sil_time_thres: 150
|
| 24 |
+
speech_2_noise_ratio: 1.0
|
| 25 |
+
do_extend: 1
|
| 26 |
+
lookback_time_start_point: 200
|
| 27 |
+
lookahead_time_end_point: 100
|
| 28 |
+
max_single_segment_time: 60000
|
| 29 |
+
snr_thres: -100.0
|
| 30 |
+
noise_frame_num_used_for_snr: 100
|
| 31 |
+
decibel_thres: -100.0
|
| 32 |
+
speech_noise_thres: 0.6
|
| 33 |
+
fe_prior_thres: 0.0001
|
| 34 |
+
silence_pdf_num: 1
|
| 35 |
+
sil_pdf_ids: [0]
|
| 36 |
+
speech_noise_thresh_low: -0.1
|
| 37 |
+
speech_noise_thresh_high: 0.3
|
| 38 |
+
output_frame_probs: False
|
| 39 |
+
frame_in_ms: 10
|
| 40 |
+
frame_length_ms: 25
|
| 41 |
+
|
| 42 |
+
encoder: FSMN
|
| 43 |
+
encoder_conf:
|
| 44 |
+
input_dim: 400
|
| 45 |
+
input_affine_dim: 140
|
| 46 |
+
fsmn_layers: 4
|
| 47 |
+
linear_dim: 250
|
| 48 |
+
proj_dim: 128
|
| 49 |
+
lorder: 20
|
| 50 |
+
rorder: 0
|
| 51 |
+
lstride: 1
|
| 52 |
+
rstride: 0
|
| 53 |
+
output_affine_dim: 140
|
| 54 |
+
output_dim: 248
|
| 55 |
+
|
| 56 |
+
|
speech_fsmn_vad/configuration.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"framework": "pytorch",
|
| 3 |
+
"task" : "voice-activity-detection",
|
| 4 |
+
"pipeline": {"type":"funasr-pipeline"},
|
| 5 |
+
"model": {"type" : "funasr"},
|
| 6 |
+
"file_path_metas": {
|
| 7 |
+
"init_param":"model.pt",
|
| 8 |
+
"config":"config.yaml",
|
| 9 |
+
"frontend_conf":{"cmvn_file": "am.mvn"}},
|
| 10 |
+
"model_name_in_hub": {
|
| 11 |
+
"ms":"iic/speech_fsmn_vad_zh-cn-16k-common-pytorch",
|
| 12 |
+
"hf":""}
|
| 13 |
+
}
|
speech_fsmn_vad/model.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b3be75be477f0780277f3bae0fe489f48718f585f3a6e45d7dd1fbb1a4255fc5
|
| 3 |
+
size 1721366
|
speech_fsmn_vad/struct.png
ADDED
|
speech_tokenizer_v3.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:23236a74175dbdda47afc66dbadd5bcb41303c467a57c261cb8539ad9db9208d
|
| 3 |
+
size 969451503
|
version-RFB-320.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:68bbbaa1023629ab4967c735c133ec2d440d94941cdcb4e0d9c9cd2ab0c83c0d
|
| 3 |
+
size 1231013
|