| 11 | |
| 12 | |
| 13 | def _parse(item, prompt_mode): |
| 14 | # 1. 从 Choices 字符串里按行拆分出每个选项 |
| 15 | raw_choices = item.get('Choices', '') |
| 16 | # 去掉首尾空白并按行分割,过滤掉空行 |
| 17 | lines = [ |
| 18 | line.strip() for line in raw_choices.strip().splitlines() |
| 19 | if line.strip() |
| 20 | ] |
| 21 | |
| 22 | # 2. 用正则去掉行首的 "A. "/"B. " 等前缀,只保留选项内容 |
| 23 | options_list = [re.sub(r'^[A-Z]\.\s*', '', line) for line in lines] |
| 24 | |
| 25 | # 3. 写回 item |
| 26 | item['options'] = options_list |
| 27 | |
| 28 | # 4. 重建带标号的选项字符串 |
| 29 | options_str = '\n'.join(f'{chr(65 + i)}. {opt}' |
| 30 | for i, opt in enumerate(options_list)) |
| 31 | |
| 32 | # 5. 构造 question、label、prompt_mode、start、end |
| 33 | item['question'] = f"{item['Question']}\n{options_str}" |
| 34 | item['label'] = item['Answer'] |
| 35 | item['prompt_mode'] = prompt_mode |
| 36 | item['start'] = chr(65) |
| 37 | item['end'] = chr(65 + len(options_list) - 1) |
| 38 | return item |
| 39 | |
| 40 | |
| 41 | @LOAD_DATASET.register_module() |