MCPcopy Create free account
hub / github.com/espnet/espnet / _add_noise

Method _add_noise

espnet2/train/preprocessor.py:309–381  ·  view source on GitHub ↗
(
        self,
        speech,
        power,
        noises,
        noise_db_low,
        noise_db_high,
        tgt_fs=None,
        single_channel=False,
    )

Source from the content-addressed store, hash-verified

307 return speech, rir
308
309 def _add_noise(
310 self,
311 speech,
312 power,
313 noises,
314 noise_db_low,
315 noise_db_high,
316 tgt_fs=None,
317 single_channel=False,
318 ):
319 nsamples = speech.shape[1]
320 noise_path = np.random.choice(noises)
321 noise = None
322 if noise_path is not None:
323 noise_db = np.random.uniform(noise_db_low, noise_db_high)
324 with soundfile.SoundFile(noise_path) as f:
325 fs = f.samplerate
326 if tgt_fs and fs != tgt_fs:
327 nsamples_ = int(nsamples / tgt_fs * fs) + 1
328 else:
329 nsamples_ = nsamples
330 if f.frames == nsamples_:
331 noise = f.read(dtype=np.float64, always_2d=True)
332 elif f.frames < nsamples_:
333 if f.frames / nsamples_ < self.short_noise_thres:
334 logging.warning(
335 f"Noise ({f.frames}) is much shorter than "
336 f"speech ({nsamples_}) in dynamic mixing"
337 )
338 offset = np.random.randint(0, nsamples_ - f.frames)
339 # noise: (Time, Nmic)
340 noise = f.read(dtype=np.float64, always_2d=True)
341 # Repeat noise
342 noise = np.pad(
343 noise,
344 [(offset, nsamples_ - f.frames - offset), (0, 0)],
345 mode="wrap",
346 )
347 else:
348 offset = np.random.randint(0, f.frames - nsamples_)
349 f.seek(offset)
350 # noise: (Time, Nmic)
351 noise = f.read(nsamples_, dtype=np.float64, always_2d=True)
352 if len(noise) != nsamples_:
353 raise RuntimeError(f"Something wrong: {noise_path}")
354 if single_channel:
355 num_ch = noise.shape[1]
356 chs = [np.random.randint(num_ch)]
357 noise = noise[:, chs]
358 # noise: (Nmic, Time)
359 noise = noise.T
360 if tgt_fs and fs != tgt_fs:
361 logging.warning(
362 f"Resampling noise to match the sampling rate ({fs} -> {tgt_fs} Hz)"
363 )
364 noise = librosa.resample(
365 noise, orig_sr=fs, target_sr=tgt_fs, res_type="kaiser_fast"
366 )

Callers 2

_speech_processMethod · 0.95
_speech_processMethod · 0.80

Calls

no outgoing calls

Tested by

no test coverage detected