nassimaODL commited on
Commit
f2fe655
·
0 Parent(s):

Duplicate from hi-paris/CosyVoice2-0.5B-EU

Browse files
.gitattributes ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ sample_audio_prompt.wav filter=lfs diff=lfs merge=lfs -text
CosyVoice-BlankEN/config.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 151643,
7
+ "eos_token_id": 151645,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 896,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 4864,
12
+ "max_position_embeddings": 32768,
13
+ "max_window_layers": 24,
14
+ "model_type": "qwen2",
15
+ "num_attention_heads": 14,
16
+ "num_hidden_layers": 24,
17
+ "num_key_value_heads": 2,
18
+ "rms_norm_eps": 1e-06,
19
+ "rope_theta": 1000000.0,
20
+ "sliding_window": 32768,
21
+ "tie_word_embeddings": true,
22
+ "torch_dtype": "bfloat16",
23
+ "transformers_version": "4.40.1",
24
+ "use_cache": true,
25
+ "use_sliding_window": false,
26
+ "vocab_size": 151936
27
+ }
CosyVoice-BlankEN/generation_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "pad_token_id": 151643,
4
+ "do_sample": true,
5
+ "eos_token_id": [
6
+ 151645,
7
+ 151643
8
+ ],
9
+ "repetition_penalty": 1.1,
10
+ "temperature": 0.7,
11
+ "top_p": 0.8,
12
+ "top_k": 20,
13
+ "transformers_version": "4.37.0"
14
+ }
CosyVoice-BlankEN/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
CosyVoice-BlankEN/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:130282af0dfa9fe5840737cc49a0d339d06075f83c5a315c3372c9a0740d0b96
3
+ size 988097824
CosyVoice-BlankEN/tokenizer_config.json ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "151643": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "151644": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "151645": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ }
28
+ },
29
+ "additional_special_tokens": ["<|im_start|>", "<|im_end|>"],
30
+ "bos_token": null,
31
+ "chat_template": "{% for message in messages %}{% if loop.first and messages[0]['role'] != 'system' %}{{ '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}{% endif %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
32
+ "clean_up_tokenization_spaces": false,
33
+ "eos_token": "<|im_end|>",
34
+ "errors": "replace",
35
+ "model_max_length": 32768,
36
+ "pad_token": "<|endoftext|>",
37
+ "split_special_tokens": false,
38
+ "tokenizer_class": "Qwen2Tokenizer",
39
+ "unk_token": null
40
+ }
CosyVoice-BlankEN/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
README.md ADDED
@@ -0,0 +1,100 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ language:
4
+ - en
5
+ - de
6
+ - fr
7
+ - zh
8
+ - ko
9
+ - ja
10
+ base_model:
11
+ - FunAudioLLM/CosyVoice2-0.5B
12
+ - Qwen/Qwen3-0.6B
13
+ - utter-project/EuroLLM-1.7B-Instruct
14
+ - mistralai/Mistral-7B-v0.3
15
+ pipeline_tag: text-to-speech
16
+ ---
17
+
18
+ <p align="center">
19
+ <img src="https://hi-paris.github.io/CosyVoice2-EU/cosyvoice2-logo-clear.png"
20
+ alt="CosyVoice2-EU logo" width="260">
21
+ </p>
22
+
23
+
24
+ # CosyVoice2-0.5B-EU — FR/DE Zero-Shot Voice Cloning (CosyVoice2)
25
+
26
+ **Europeanized CosyVoice2 for French & German.**
27
+ Plug-and-play zero-shot voice cloning with streaming support, bilingual training (FR+DE), and a simple CLI via the companion PyPI package.
28
+
29
+ **👉 PyPI:** `cosyvoice2-eu` (current: **0.2.7**) at https://pypi.org/project/cosyvoice2-eu/
30
+ **👉 Demo:** https://hi-paris.github.io/CosyVoice2-EU/
31
+ **👉 Built on:** FunAudioLLM **CosyVoice2** (semantic LM + chunk-aware flow + HiFi-GAN)
32
+
33
+ ---
34
+
35
+ ## TL;DR
36
+ High-quality **French/German** zero-shot TTS (text + short reference audio) built on **CosyVoice2**. Optimized for sentence-to-paragraph narration, bilingual FR+DE adaptation, and easy local inference.
37
+ While this model is optimized for French and German, it remains fully compatible with the original CosyVoice2 languages — English, Chinese, Japanese, Korean, and their dialects.
38
+
39
+ ---
40
+
41
+ ## Quickstart (CLI)
42
+
43
+ Install:
44
+ ```bash
45
+ pip install cosyvoice2-eu
46
+ ```
47
+
48
+ French example:
49
+ ```bash
50
+ cosy2-eu --text "Salut ! Je vous présente CosyVoice 2, un système de synthèse vocale très avancé." --prompt path/to/french_ref.wav --out out_fr.wav
51
+ ```
52
+
53
+ German example:
54
+ ```bash
55
+ cosy2-eu --text "Hallo! Ich präsentiere CosyVoice 2 – ein fortschrittliches TTS-System." --prompt path/to/german_ref.wav --out out_de.wav
56
+ ```
57
+
58
+ > First run downloads the model from this repo and caches it locally.
59
+ > Tip: You can experiment with prompts for style control using `"<style>. <|endofprompt|> <text>"`, e.g., "Speak cheerfully. <|endofprompt|> Hallo! Wie geht es Ihnen heute?"
60
+
61
+ ---
62
+
63
+ ## What you get
64
+ - **Zero-shot voice cloning** for **FR/DE** (reference audio → cloned timbre & style).
65
+ - **Bilingual adaptation** (FR+DE) on top of CosyVoice2 for stronger data efficiency. While this model adds support for French and German, it remains fully compatible with the original CosyVoice2 languages — English, Chinese, Japanese, Korean, and their dialects.
66
+ - **Streaming & non-streaming** synthesis supported by the underlying architecture.
67
+ - **Simple local inference**: one pip install, one CLI (`cosy2-eu`).
68
+ - **Interoperable components** (text→semantic LM, flow decoder, HiFi-GAN vocoder).
69
+
70
+ Also compatible with original CosyVoice2 languages (EN/ZH/JA/KO & dialects).
71
+
72
+ ---
73
+
74
+ ## Inputs / Outputs
75
+ - **Input:** text (FR/DE) + short **reference audio** (mono WAV recommended).
76
+ - **Output:** synthesized WAV cloning the reference speaker’s timbre, speaking the input text in FR/DE.
77
+
78
+ ---
79
+
80
+ ## Notes & limitations
81
+ - FR/DE were adapted under constrained open-data budgets; extreme edge cases (very noisy prompts, long numerics, heavy code-switching) may require careful prompting or additional fine-tuning.
82
+ - Voice cloning carries **misuse risks** (impersonation, fraud). Use only with consent and follow local laws/policies.
83
+
84
+ ---
85
+
86
+ ## License & attribution
87
+ - **License:** Apache-2.0 (see card metadata / repo).
88
+ - Built on **CosyVoice2** by FunAudioLLM; please cite their work (see below).
89
+
90
+
91
+ ---
92
+
93
+ **Links**
94
+ - PyPI (inference CLI): https://pypi.org/project/cosyvoice2-eu/
95
+ - Upstream project: https://github.com/FunAudioLLM/CosyVoice
96
+ - CosyVoice2 paper & page: https://arxiv.org/abs/2412.10117 • https://funaudiollm.github.io/cosyvoice2/
97
+
98
+ ---
99
+
100
+ *If you use CosyVoice2-0.5B-EU in research or products, please add a short acknowledgment and share feedback or samples—we’re continuously improving FR/DE expressiveness and robustness.*
campplus.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a6ac6a63997761ae2997373e2ee1c47040854b4b759ea41ec48e4e42df0f4d73
3
+ size 28303423
cosyvoice2.yaml ADDED
@@ -0,0 +1,237 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # set random seed, so that you may reproduce your result.
2
+ __set_seed1: !apply:random.seed [1986]
3
+ __set_seed2: !apply:numpy.random.seed [1986]
4
+ __set_seed3: !apply:torch.manual_seed [1986]
5
+ __set_seed4: !apply:torch.cuda.manual_seed_all [1986]
6
+
7
+ # fixed params
8
+ sample_rate: 24000
9
+ llm_input_size: 896
10
+ llm_output_size: 896
11
+ spk_embed_dim: 192
12
+ qwen_pretrain_path: ''
13
+ token_frame_rate: 25
14
+ token_mel_ratio: 2
15
+
16
+ # stream related params
17
+ chunk_size: 25 # streaming inference chunk size, in token
18
+ num_decoding_left_chunks: -1 # streaming inference flow decoder left chunk size, <0 means use all left chunks
19
+
20
+ # model params
21
+ # for all class/function included in this repo, we use !<name> or !<new> for intialization, so that user may find all corresponding class/function according to one single yaml.
22
+ # for system/third_party class/function, we do not require this.
23
+ llm: !new:cosyvoice.llm.llm.Qwen2LM
24
+ # llm_input_size/llm_output_size will be auto-set from backbone.hidden_size, so 896 here is ignored
25
+ llm_input_size: 0
26
+ llm_output_size: 0
27
+ speech_token_size: 6561
28
+ length_normalized_loss: True
29
+ lsm_weight: 0
30
+ mix_ratio: [5, 15]
31
+ llm: !new:cosyvoice.llm.llm.HFBackbone # backbone-agnostic
32
+ pretrain_path: !ref <qwen_pretrain_path> # e.g., "Qwen/Qwen3-0.6B" or "mistralai/Mistral-7B-Instruct-v0.3"
33
+ sampling: !name:cosyvoice.utils.common.ras_sampling
34
+ top_p: 0.8
35
+ top_k: 25
36
+ win_size: 10
37
+ tau_r: 0.1
38
+
39
+ flow: !new:cosyvoice.flow.flow.CausalMaskedDiffWithXvec
40
+ input_size: 512
41
+ output_size: 80
42
+ spk_embed_dim: !ref <spk_embed_dim>
43
+ output_type: 'mel'
44
+ vocab_size: 6561
45
+ input_frame_rate: !ref <token_frame_rate>
46
+ only_mask_loss: True
47
+ token_mel_ratio: !ref <token_mel_ratio>
48
+ pre_lookahead_len: 3
49
+ encoder: !new:cosyvoice.transformer.upsample_encoder.UpsampleConformerEncoder
50
+ output_size: 512
51
+ attention_heads: 8
52
+ linear_units: 2048
53
+ num_blocks: 6
54
+ dropout_rate: 0.1
55
+ positional_dropout_rate: 0.1
56
+ attention_dropout_rate: 0.1
57
+ normalize_before: True
58
+ input_layer: 'linear'
59
+ pos_enc_layer_type: 'rel_pos_espnet'
60
+ selfattention_layer_type: 'rel_selfattn'
61
+ input_size: 512
62
+ use_cnn_module: False
63
+ macaron_style: False
64
+ static_chunk_size: !ref <chunk_size>
65
+ decoder: !new:cosyvoice.flow.flow_matching.CausalConditionalCFM
66
+ in_channels: 240
67
+ n_spks: 1
68
+ spk_emb_dim: 80
69
+ cfm_params: !new:omegaconf.DictConfig
70
+ content:
71
+ sigma_min: 1e-06
72
+ solver: 'euler'
73
+ t_scheduler: 'cosine'
74
+ training_cfg_rate: 0.2
75
+ inference_cfg_rate: 0.7
76
+ reg_loss_type: 'l1'
77
+ estimator: !new:cosyvoice.flow.decoder.CausalConditionalDecoder
78
+ in_channels: 320
79
+ out_channels: 80
80
+ channels: [256]
81
+ dropout: 0.0
82
+ attention_head_dim: 64
83
+ n_blocks: 4
84
+ num_mid_blocks: 12
85
+ num_heads: 8
86
+ act_fn: 'gelu'
87
+ static_chunk_size: !ref <chunk_size> * <token_mel_ratio>
88
+ num_decoding_left_chunks: !ref <num_decoding_left_chunks>
89
+
90
+ hift: !new:cosyvoice.hifigan.generator.HiFTGenerator
91
+ in_channels: 80
92
+ base_channels: 512
93
+ nb_harmonics: 8
94
+ sampling_rate: !ref <sample_rate>
95
+ nsf_alpha: 0.1
96
+ nsf_sigma: 0.003
97
+ nsf_voiced_threshold: 10
98
+ upsample_rates: [8, 5, 3]
99
+ upsample_kernel_sizes: [16, 11, 7]
100
+ istft_params:
101
+ n_fft: 16
102
+ hop_len: 4
103
+ resblock_kernel_sizes: [3, 7, 11]
104
+ resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
105
+ source_resblock_kernel_sizes: [7, 7, 11]
106
+ source_resblock_dilation_sizes: [[1, 3, 5], [1, 3, 5], [1, 3, 5]]
107
+ lrelu_slope: 0.1
108
+ audio_limit: 0.99
109
+ f0_predictor: !new:cosyvoice.hifigan.f0_predictor.ConvRNNF0Predictor
110
+ num_class: 1
111
+ in_channels: 80
112
+ cond_channels: 512
113
+
114
+ # gan related module
115
+ mel_spec_transform1: !name:matcha.utils.audio.mel_spectrogram
116
+ n_fft: 1920
117
+ num_mels: 80
118
+ sampling_rate: !ref <sample_rate>
119
+ hop_size: 480
120
+ win_size: 1920
121
+ fmin: 0
122
+ fmax: null
123
+ center: False
124
+ hifigan: !new:cosyvoice.hifigan.hifigan.HiFiGan
125
+ generator: !ref <hift>
126
+ discriminator: !new:cosyvoice.hifigan.discriminator.MultipleDiscriminator
127
+ mpd: !new:matcha.hifigan.models.MultiPeriodDiscriminator
128
+ mrd: !new:cosyvoice.hifigan.discriminator.MultiResSpecDiscriminator
129
+ mel_spec_transform: [
130
+ !ref <mel_spec_transform1>
131
+ ]
132
+
133
+ # processor functions
134
+ parquet_opener: !name:cosyvoice.dataset.processor.parquet_opener
135
+ get_tokenizer: !name:cosyvoice.tokenizer.tokenizer.get_qwen_tokenizer
136
+ token_path: !ref <qwen_pretrain_path>
137
+ skip_special_tokens: True
138
+ # add_additional_specials: auto-detected based on token_path (True for blanken/CosyVoice models, False for custom HF backbones)
139
+ allowed_special: 'all'
140
+ tokenize: !name:cosyvoice.dataset.processor.tokenize
141
+ get_tokenizer: !ref <get_tokenizer>
142
+ allowed_special: !ref <allowed_special>
143
+ filter: !name:cosyvoice.dataset.processor.filter
144
+ max_length: 40960
145
+ min_length: 100
146
+ token_max_length: 512 # not sure if this can just be changed?
147
+ token_min_length: 1
148
+ resample: !name:cosyvoice.dataset.processor.resample
149
+ resample_rate: !ref <sample_rate>
150
+ truncate: !name:cosyvoice.dataset.processor.truncate
151
+ truncate_length: 24480 # must be a multiplier of hop_size
152
+ feat_extractor: !name:matcha.utils.audio.mel_spectrogram
153
+ n_fft: 1920
154
+ num_mels: 80
155
+ sampling_rate: !ref <sample_rate>
156
+ hop_size: 480
157
+ win_size: 1920
158
+ fmin: 0
159
+ fmax: 8000
160
+ center: False
161
+ compute_fbank: !name:cosyvoice.dataset.processor.compute_fbank
162
+ feat_extractor: !ref <feat_extractor>
163
+ token_mel_ratio: 2
164
+ compute_f0: !name:cosyvoice.dataset.processor.compute_f0
165
+ sample_rate: !ref <sample_rate>
166
+ hop_size: 480
167
+ parse_embedding: !name:cosyvoice.dataset.processor.parse_embedding
168
+ normalize: True
169
+ shuffle: !name:cosyvoice.dataset.processor.shuffle
170
+ shuffle_size: 1000
171
+ sort: !name:cosyvoice.dataset.processor.sort
172
+ sort_size: 500 # sort_size should be less than shuffle_size
173
+ batch: !name:cosyvoice.dataset.processor.batch
174
+ batch_type: 'dynamic'
175
+ max_frames_in_batch: 3000
176
+ padding: !name:cosyvoice.dataset.processor.padding
177
+ use_spk_embedding: True # change to True during sft
178
+
179
+
180
+ # dataset processor pipeline
181
+ data_pipeline: [
182
+ !ref <parquet_opener>,
183
+ !ref <tokenize>,
184
+ !ref <filter>,
185
+ !ref <resample>,
186
+ !ref <compute_fbank>,
187
+ !ref <parse_embedding>,
188
+ !ref <shuffle>,
189
+ !ref <sort>,
190
+ !ref <batch>,
191
+ !ref <padding>,
192
+ ]
193
+ data_pipeline_gan: [
194
+ !ref <parquet_opener>,
195
+ !ref <tokenize>,
196
+ !ref <filter>,
197
+ !ref <resample>,
198
+ !ref <truncate>,
199
+ !ref <compute_fbank>,
200
+ !ref <compute_f0>,
201
+ !ref <parse_embedding>,
202
+ !ref <shuffle>,
203
+ !ref <sort>,
204
+ !ref <batch>,
205
+ !ref <padding>,
206
+ ]
207
+
208
+ # llm flow train conf
209
+ train_conf:
210
+ optim: adamw
211
+ optim_conf:
212
+ lr: 1e-5 # change to 1e-5 during sft
213
+ # weight_decay: 0.01
214
+ scheduler: constantlr # change to constantlr during sft
215
+ scheduler_conf:
216
+ warmup_steps: 2500
217
+ max_epoch: 30 # 200
218
+ grad_clip: 5
219
+ accum_grad: 4
220
+ log_interval: 100
221
+ save_per_step: 3000 # -1 this is where you can set the step-wise validation checkpoint, -1 means no step-wise validation checkpoint
222
+
223
+ # gan train conf
224
+ train_conf_gan:
225
+ optim: adam
226
+ optim_conf:
227
+ lr: 0.0002 # use small lr for gan training
228
+ scheduler: constantlr
229
+ optim_d: adam
230
+ optim_conf_d:
231
+ lr: 0.0002 # use small lr for gan training
232
+ scheduler_d: constantlr
233
+ max_epoch: 20
234
+ grad_clip: 5
235
+ accum_grad: 1 # in gan training, accum_grad must be 1
236
+ log_interval: 100
237
+ save_per_step: -1
flow.decoder.estimator.fp32.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:628863056c6462e7563c3ce0c04f3434b8d0eabd56cc1ccc58edf58ceb0d758f
3
+ size 286312346
flow.encoder.fp16.zip ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:308e0517a8c35e36d66f25e7e467dd0f57fadb7c5f6c1a5f743e9e82db53b559
3
+ size 116706755
flow.encoder.fp32.zip ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2819dc8568189a4dd3792669b2e57ac1be183be653b2f5cf97ad11031c17b7ef
3
+ size 192369091
flow.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c51909e408ab4b9ab57179f99ce1c97a7f8f4618fccd68f067a1d9de85783062
3
+ size 450569991
hifigan.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:04936834f1bdcfeb0973203bb53b9f8ea652b779a06f3ff3958ee7693d7bef96
3
+ size 248993278
hift.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3386cc880324d4e98e05987b99107f49e40ed925b8ecc87c1f4939432d429879
3
+ size 83390254
llm.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eaa230b528e30f7684dfa2bbc984ccbc336299c3d475f9722224a60d476f2c3a
3
+ size 2567851323
sample_audio_prompt.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b4e3ab32edbbc2561141ccd12636be2c1bc304a6cbd8696d69213246dd4cf1e
3
+ size 5690588
speech_tokenizer_v2.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d43342aa12163a80bf07bffb94c9de2e120a8df2f9917cd2f642e7f4219c6f71
3
+ size 496082973