File size: 5,183 Bytes
af7b5a0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
token_list:
- <blank>
- <unk>
- a
- o
- i
- '['
- '#'
- u
- ']'
- e
- k
- n
- t
- r
- s
- N
- m
- _
- sh
- d
- g
- ^
- $
- w
- cl
- h
- y
- b
- j
- ts
- ch
- z
- p
- f
- ky
- ry
- gy
- hy
- ny
- by
- my
- py
- v
- dy
- '?'
- ty
- <sos/eos>
odim: null
model_conf: {}
use_preprocessor: true
token_type: phn
bpemodel: null
non_linguistic_symbols: null
cleaner: jaconv
g2p: pyopenjtalk_prosody
feats_extract: linear_spectrogram
feats_extract_conf:
    n_fft: 2048
    hop_length: 512
    win_length: null
normalize: null
normalize_conf: {}
tts: vits
tts_conf:
    generator_type: vits_generator
    generator_params:
        hidden_channels: 192
        spks: -1
        global_channels: -1
        segment_size: 32
        text_encoder_attention_heads: 2
        text_encoder_ffn_expand: 4
        text_encoder_blocks: 6
        text_encoder_positionwise_layer_type: conv1d
        text_encoder_positionwise_conv_kernel_size: 3
        text_encoder_positional_encoding_layer_type: rel_pos
        text_encoder_self_attention_layer_type: rel_selfattn
        text_encoder_activation_type: swish
        text_encoder_normalize_before: true
        text_encoder_dropout_rate: 0.1
        text_encoder_positional_dropout_rate: 0.0
        text_encoder_attention_dropout_rate: 0.1
        use_macaron_style_in_text_encoder: true
        use_conformer_conv_in_text_encoder: false
        text_encoder_conformer_kernel_size: -1
        decoder_kernel_size: 7
        decoder_channels: 512
        decoder_upsample_scales:
        - 8
        - 8
        - 2
        - 2
        - 2
        decoder_upsample_kernel_sizes:
        - 16
        - 16
        - 4
        - 4
        - 4
        decoder_resblock_kernel_sizes:
        - 3
        - 7
        - 11
        decoder_resblock_dilations:
        -   - 1
            - 3
            - 5
        -   - 1
            - 3
            - 5
        -   - 1
            - 3
            - 5
        use_weight_norm_in_decoder: true
        posterior_encoder_kernel_size: 5
        posterior_encoder_layers: 16
        posterior_encoder_stacks: 1
        posterior_encoder_base_dilation: 1
        posterior_encoder_dropout_rate: 0.0
        use_weight_norm_in_posterior_encoder: true
        flow_flows: 4
        flow_kernel_size: 5
        flow_base_dilation: 1
        flow_layers: 4
        flow_dropout_rate: 0.0
        use_weight_norm_in_flow: true
        use_only_mean_in_flow: true
        stochastic_duration_predictor_kernel_size: 3
        stochastic_duration_predictor_dropout_rate: 0.5
        stochastic_duration_predictor_flows: 4
        stochastic_duration_predictor_dds_conv_layers: 3
        vocabs: 47
        aux_channels: 1025
    discriminator_type: hifigan_multi_scale_multi_period_discriminator
    discriminator_params:
        scales: 1
        scale_downsample_pooling: AvgPool1d
        scale_downsample_pooling_params:
            kernel_size: 4
            stride: 2
            padding: 2
        scale_discriminator_params:
            in_channels: 1
            out_channels: 1
            kernel_sizes:
            - 15
            - 41
            - 5
            - 3
            channels: 128
            max_downsample_channels: 1024
            max_groups: 16
            bias: true
            downsample_scales:
            - 2
            - 2
            - 4
            - 4
            - 1
            nonlinear_activation: LeakyReLU
            nonlinear_activation_params:
                negative_slope: 0.1
            use_weight_norm: true
            use_spectral_norm: false
        follow_official_norm: false
        periods:
        - 2
        - 3
        - 5
        - 7
        - 11
        period_discriminator_params:
            in_channels: 1
            out_channels: 1
            kernel_sizes:
            - 5
            - 3
            channels: 32
            downsample_scales:
            - 3
            - 3
            - 3
            - 3
            - 1
            max_downsample_channels: 1024
            bias: true
            nonlinear_activation: LeakyReLU
            nonlinear_activation_params:
                negative_slope: 0.1
            use_weight_norm: true
            use_spectral_norm: false
    generator_adv_loss_params:
        average_by_discriminators: false
        loss_type: mse
    discriminator_adv_loss_params:
        average_by_discriminators: false
        loss_type: mse
    feat_match_loss_params:
        average_by_discriminators: false
        average_by_layers: false
        include_final_outputs: true
    mel_loss_params:
        fs: 44100
        n_fft: 2048
        hop_length: 512
        win_length: null
        window: hann
        n_mels: 80
        fmin: 0
        fmax: null
        log_base: null
    lambda_adv: 1.0
    lambda_mel: 45.0
    lambda_feat_match: 2.0
    lambda_dur: 1.0
    lambda_kl: 1.0
    sampling_rate: 44100
    cache_generator_outputs: true
pitch_extract: null
pitch_extract_conf: {}
pitch_normalize: null
pitch_normalize_conf: {}
energy_extract: null
energy_extract_conf: {}
energy_normalize: null
energy_normalize_conf: {}
required:
- output_dir
- token_list
version: '202308'
distributed: false