File size: 6,211 Bytes
affcd23
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
{
    "name": "default_config",
    "n_gpu": 1,
    "text_encoder": {
        "type": "CTCCharTextEncoder",
        "args": {
            "kenlm_model_path": "hw_asr/text_encoder/lower_3-gram.arpa",
            "unigrams_path": "hw_asr/text_encoder/librispeech-fixed-vocab.txt"
        }
    },
    "preprocessing": {
        "sr": 16000,
        "spectrogram": {
            "type": "MelSpectrogram",
            "args": {
                "n_mels": 256
            }
        },
        "log_spec": true
    },
    "augmentations": {
        "random_apply_p": 0.6,
        "wave": [
            {
                "type": "AddColoredNoise",
                "args": {
                    "p": 1,
                    "sample_rate": 16000
                }
            },
            {
                "type": "Gain",
                "args": {
                    "p": 0.8,
                    "sample_rate": 16000
                }
            },
            {
                "type": "HighPassFilter",
                "args": {
                    "p": 0,
                    "sample_rate": 16000
                }
            },
            {
                "type": "LowPassFilter",
                "args": {
                    "p": 0,
                    "sample_rate": 16000
                }
            },
            {
                "type": "PitchShift",
                "args": {
                    "p": 0.8,
                    "min_transpose_semitones": -2,
                    "max_transpose_semitones": 2,
                    "sample_rate": 16000
                }
            },
            {
                "type": "PolarityInversion",
                "args": {
                    "p": 0.8,
                    "sample_rate": 16000
                }
            },
            {
                "type": "Shift",
                "args": {
                    "p": 0.8,
                    "sample_rate": 16000
                }
            }
        ],
        "spectrogram": [
            {
                "type": "TimeMasking",
                "args": {
                    "time_mask_param": 80,
                    "p": 0.05
                }
            },
            {
                "type": "FrequencyMasking",
                "args": {
                    "freq_mask_param": 80
                }
            }
        ]
    },
    "arch": {
        "type": "DeepSpeech2Model",
        "args": {
            "n_feats": 256,
            "n_rnn_layers": 6,
            "rnn_hidden_size": 512,
            "rnn_dropout": 0.2
        }
    },
    "data": {
        "train": {
            "batch_size": 128,
            "num_workers": 4,
            "datasets": [
                {
                    "type": "LibrispeechDataset",
                    "args": {
                        "part": "train-clean-100",
                        "max_audio_length": 40.0,
                        "max_text_length": 400
                    }
                },
                {
                    "type": "LibrispeechDataset",
                    "args": {
                        "part": "train-clean-360",
                        "max_audio_length": 40.0,
                        "max_text_length": 400
                    }
                },
                {
                    "type": "LibrispeechDataset",
                    "args": {
                        "part": "train-other-500",
                        "max_audio_length": 40.0,
                        "max_text_length": 400
                    }
                }
            ]
        },
        "val": {
            "batch_size": 64,
            "num_workers": 4,
            "datasets": [
                {
                    "type": "LibrispeechDataset",
                    "args": {
                        "part": "dev-clean"
                    }
                }
            ]
        },
        "test-other": {
            "batch_size": 64,
            "num_workers": 4,
            "datasets": [
                {
                    "type": "LibrispeechDataset",
                    "args": {
                        "part": "test-other"
                    }
                }
            ]
        },
        "test-clean": {
            "batch_size": 64,
            "num_workers": 4,
            "datasets": [
                {
                    "type": "LibrispeechDataset",
                    "args": {
                        "part": "test-clean"
                    }
                }
            ]
        }
    },
    "optimizer": {
        "type": "AdamW",
        "args": {
            "lr": 0.0003,
            "weight_decay": 1e-05
        }
    },
    "loss": {
        "type": "CTCLoss",
        "args": {}
    },
    "metrics": [
        {
            "type": "ArgmaxWERMetric",
            "args": {
                "name": "WER (argmax)"
            }
        },
        {
            "type": "ArgmaxCERMetric",
            "args": {
                "name": "CER (argmax)"
            }
        },
        {
            "type": "BeamSearchWERMetric",
            "args": {
                "beam_size": 4,
                "name": "WER (beam search)"
            }
        },
        {
            "type": "BeamSearchCERMetric",
            "args": {
                "beam_size": 4,
                "name": "CER (beam search)"
            }
        },
        {
            "type": "LanguageModelWERMetric",
            "args": {
                "name": "WER (LM)"
            }
        },
        {
            "type": "LanguageModelCERMetric",
            "args": {
                "name": "CER (LM)"
            }
        }
    ],
    "lr_scheduler": {
        "type": "OneCycleLR",
        "args": {
            "steps_per_epoch": 1000,
            "epochs": 50,
            "anneal_strategy": "cos",
            "max_lr": 0.0003,
            "pct_start": 0.1
        }
    },
    "trainer": {
        "epochs": 50,
        "save_dir": "saved/",
        "save_period": 5,
        "verbosity": 2,
        "monitor": "min val_loss",
        "early_stop": 100,
        "visualize": "wandb",
        "wandb_project": "asr_project",
        "len_epoch": 1000,
        "grad_norm_clip": 10
    }
}