Load Kincaid46 audio files and transcripts. Parses the CSV metadata file and maps entries to corresponding M4A audio files using a zero-padded naming convention. Returns: Tuple[list, list]: A tuple containing: - List of M4A audio file paths
(self)
| 824 | """ |
| 825 | |
| 826 | def load(self) -> Tuple[list, list]: |
| 827 | """Load Kincaid46 audio files and transcripts. |
| 828 | |
| 829 | Parses the CSV metadata file and maps entries to corresponding M4A audio |
| 830 | files using a zero-padded naming convention. |
| 831 | |
| 832 | Returns: |
| 833 | Tuple[list, list]: A tuple containing: |
| 834 | - List of M4A audio file paths |
| 835 | - List of corresponding transcript strings from CSV column 5 |
| 836 | """ |
| 837 | audio_files, transcript_texts = [], [] |
| 838 | |
| 839 | with open(f"{self.root_dir}/text.csv", "r") as f: |
| 840 | reader = csv.reader(f) |
| 841 | for i, row in enumerate(reader): |
| 842 | if i == 0: |
| 843 | continue |
| 844 | audio_file = os.path.join(self.root_dir, "audio", f"{(i - 1):02}.m4a") |
| 845 | transcript_texts.append(row[5]) |
| 846 | audio_files.append(audio_file) |
| 847 | |
| 848 | return audio_files, transcript_texts |
| 849 | |
| 850 | |
| 851 | class CORAALLongLoader(BaseDatasetLoader): |
nothing calls this directly
no outgoing calls
no test coverage detected