(waveform, config_toml='configs/rnnt.toml')
| 7 | |
| 8 | |
| 9 | def audio_processing(waveform, config_toml='configs/rnnt.toml'): |
| 10 | config = toml.load(config_toml) |
| 11 | |
| 12 | featurizer_config = config['input_eval'] |
| 13 | audio_preprocessor = AudioPreprocessing(**featurizer_config) |
| 14 | audio_preprocessor.eval() |
| 15 | |
| 16 | assert waveform.ndim == 1 |
| 17 | waveform_length = np.array(waveform.shape[0], dtype=np.int64) |
| 18 | waveform = np.expand_dims(waveform, 0) |
| 19 | waveform_length = np.expand_dims(waveform_length, 0) |
| 20 | with torch.no_grad(): |
| 21 | waveform = torch.from_numpy(waveform) |
| 22 | waveform_length = torch.from_numpy(waveform_length) |
| 23 | feature, feature_length = audio_preprocessor.forward( |
| 24 | (waveform, waveform_length)) |
| 25 | assert feature.ndim == 3 |
| 26 | assert feature_length.ndim == 1 |
| 27 | feature = feature.permute(2, 0, 1) |
| 28 | |
| 29 | return feature, feature_length |
| 30 | |
| 31 | |
| 32 | def librespeech_huggingface(config_toml='configs/rnnt.toml'): |
no test coverage detected