Plachta commited on
Commit
93e19da
·
verified ·
1 Parent(s): f2fe25c

Delete config_dit_mel_seed_facodec_small_wavenet_f0_44k.yml

Browse files
config_dit_mel_seed_facodec_small_wavenet_f0_44k.yml DELETED
@@ -1,95 +0,0 @@
1
- log_dir: ""
2
- save_freq: 1
3
- log_interval: 10
4
- save_interval: 1000
5
- device: "cuda"
6
- epochs: 1000 # number of epochs for first stage training (pre-training)
7
- batch_size: 2
8
- batch_length: 100 # maximum duration of audio in a batch (in seconds)
9
- max_len: 80 # maximum number of frames
10
- pretrained_model: ""
11
- pretrained_encoder: ""
12
- load_only_params: False # set to true if do not want to load epoch numbers and optimizer parameters
13
-
14
- preprocess_params:
15
- sr: 44100
16
- spect_params:
17
- n_fft: 2048
18
- win_length: 2048
19
- hop_length: 512
20
- n_mels: 128
21
- fmin: 0
22
- fmax: "None"
23
-
24
- model_params:
25
- dit_type: "DiT" # uDiT or DiT
26
- reg_loss_type: "l1" # l1 or l2
27
-
28
- speech_tokenizer:
29
- type: 'facodec'
30
- path: "speech_tokenizer_v1.onnx"
31
-
32
- cosyvoice:
33
- path: "../CosyVoice/pretrained_models/CosyVoice-300M"
34
-
35
- style_encoder:
36
- dim: 192
37
- campplus_path: "campplus_cn_common.bin"
38
-
39
- DAC:
40
- encoder_dim: 64
41
- encoder_rates: [2, 5, 5, 6]
42
- decoder_dim: 1536
43
- decoder_rates: [ 6, 5, 5, 2 ]
44
- sr: 24000
45
-
46
- length_regulator:
47
- channels: 512
48
- is_discrete: true
49
- content_codebook_size: 1024
50
- in_frame_rate: 80
51
- out_frame_rate: 80
52
- sampling_ratios: [1, 1, 1, 1]
53
- token_dropout_prob: 0.3 # probability of performing token dropout
54
- token_dropout_range: 1.0 # maximum percentage of tokens to drop out
55
- n_codebooks: 3
56
- quantizer_dropout: 0.5
57
- f0_condition: true
58
- n_f0_bins: 512
59
-
60
- DiT:
61
- hidden_dim: 512
62
- num_heads: 8
63
- depth: 13
64
- class_dropout_prob: 0.1
65
- block_size: 8192
66
- in_channels: 128
67
- style_condition: true
68
- final_layer_type: 'wavenet'
69
- target: 'mel' # mel or codec
70
- content_dim: 512
71
- content_codebook_size: 1024
72
- content_type: 'discrete'
73
- f0_condition: true
74
- n_f0_bins: 512
75
- content_codebooks: 1
76
- is_causal: false
77
- long_skip_connection: true
78
- zero_prompt_speech_token: false # for prompt component, do not input corresponding speech token
79
- time_as_token: false
80
- style_as_token: false
81
- uvit_skip_connection: true
82
- add_resblock_in_transformer: false
83
-
84
- wavenet:
85
- hidden_dim: 512
86
- num_layers: 8
87
- kernel_size: 5
88
- dilation_rate: 1
89
- p_dropout: 0.2
90
- style_condition: true
91
-
92
- loss_params:
93
- base_lr: 0.0001
94
- lambda_mel: 45
95
- lambda_kl: 1.0