MCPcopy Create free account
hub / github.com/espnet/espnet / _load_noise

Method _load_noise

espnet2/train/preprocessor.py:2146–2179  ·  view source on GitHub ↗
(self, speech, speech_db, noises, noise_db_low, noise_db_high)

Source from the content-addressed store, hash-verified

2144 return speech, rir
2145
2146 def _load_noise(self, speech, speech_db, noises, noise_db_low, noise_db_high):
2147 nsamples = speech.shape[1]
2148 noise_path = np.random.choice(noises)
2149 noise = None
2150 if noise_path is not None:
2151 noise_snr = np.random.uniform(noise_db_low, noise_db_high)
2152 with soundfile.SoundFile(noise_path) as f:
2153 if f.frames == nsamples:
2154 noise = f.read(dtype=np.float64)
2155 elif f.frames < nsamples:
2156 # noise: (Time,)
2157 noise = f.read(dtype=np.float64)
2158 # Repeat noise
2159 noise = np.pad(
2160 noise,
2161 (0, nsamples - f.frames),
2162 mode="wrap",
2163 )
2164 else:
2165 offset = np.random.randint(0, f.frames - nsamples)
2166 f.seek(offset)
2167 # noise: (Time,)
2168 noise = f.read(nsamples, dtype=np.float64)
2169 if len(noise) != nsamples:
2170 raise RuntimeError(f"Something wrong: {noise_path}")
2171 # noise: (Nmic, Time)
2172 noise = noise[None, :]
2173
2174 noise_power = np.mean(noise**2)
2175 noise_db = 10 * np.log10(noise_power + 1e-4)
2176 scale = np.sqrt(10 ** ((speech_db - noise_db - noise_snr) / 10))
2177
2178 noise = noise * scale
2179 return noise
2180
2181 def _apply_data_augmentation(self, speech):
2182 # speech: (Nmic, Time)

Callers 1

Calls

no outgoing calls

Tested by

no test coverage detected