MCPcopy Create free account
hub / github.com/allenai/OLMoASR / load

Method load

scripts/eval/eval.py:339–371  ·  view source on GitHub ↗

Load LibriSpeech audio files and transcripts. Recursively searches for transcript files and maps them to corresponding FLAC audio files using the LibriSpeech naming convention. Returns: Tuple[list, list]: A tuple containing: - List of audio file

(self)

Source from the content-addressed store, hash-verified

337 """
338
339 def load(self) -> Tuple[list, list]:
340 """Load LibriSpeech audio files and transcripts.
341
342 Recursively searches for transcript files and maps them to corresponding
343 FLAC audio files using the LibriSpeech naming convention.
344
345 Returns:
346 Tuple[list, list]: A tuple containing:
347 - List of audio file paths (FLAC format)
348 - List of corresponding transcript strings
349 """
350 transcript_files = []
351 audio_text = {}
352
353 for root, _, files in os.walk(self.root_dir):
354 transcript_files.extend(
355 os.path.join(root, file) for file in files if file.endswith(".txt")
356 )
357
358 for file in sorted(transcript_files):
359 with open(file, "r") as f:
360 for line in f:
361 parts = line.split(" ")
362 audio_codes = parts[0].split("-")
363 audio_file = os.path.join(
364 self.root_dir,
365 audio_codes[0],
366 audio_codes[1],
367 f"{audio_codes[0]}-{audio_codes[1]}-{audio_codes[2]}.flac",
368 )
369 audio_text[audio_file] = " ".join(parts[1:]).strip()
370
371 return list(audio_text.keys()), list(audio_text.values())
372
373
374class ArtieBiasCorpusLoader(BaseDatasetLoader):

Callers

nothing calls this directly

Calls

no outgoing calls

Tested by

no test coverage detected