(pred_str, ans,
conclusion_patterns=['CONCLUSION:'],
verbose=False,
finish_patterns=["### Reason", "Let's think step by step again", "let's go back and check", "###"],
reformat_gold_conditions=None)
| 12 | |
| 13 | |
| 14 | def parse_cot_eval(pred_str, ans, |
| 15 | conclusion_patterns=['CONCLUSION:'], |
| 16 | verbose=False, |
| 17 | finish_patterns=["### Reason", "Let's think step by step again", "let's go back and check", "###"], |
| 18 | reformat_gold_conditions=None): |
| 19 | |
| 20 | def judge_string(input_str, reformat_gold_conditions, wrong_reason, finish_patterns): |
| 21 | correct_count = 0 |
| 22 | is_correct = False |
| 23 | beyond_id = len(reformat_gold_conditions)+1 |
| 24 | beyond_id_pattern = f"({beyond_id})" |
| 25 | |
| 26 | for finish_pattern in finish_patterns: |
| 27 | if finish_pattern in input_str: |
| 28 | input_str = input_str.split(finish_pattern)[0] |
| 29 | |
| 30 | if beyond_id_pattern in input_str: |
| 31 | is_correct = False |
| 32 | wrong_reason = "beyond_list" |
| 33 | elif "if" in input_str: |
| 34 | is_correct = False |
| 35 | wrong_reason = "contain_if" |
| 36 | else: |
| 37 | is_correct = True |
| 38 | for gold_condition in reformat_gold_conditions: |
| 39 | if gold_condition not in input_str: |
| 40 | is_correct = False |
| 41 | wrong_reason = "wrong_identity" |
| 42 | else: |
| 43 | correct_count += 1 |
| 44 | correct_ratio = correct_count/len(reformat_gold_conditions) |
| 45 | |
| 46 | return is_correct, wrong_reason, correct_ratio |
| 47 | |
| 48 | def check_numbers_in_string(s, N): |
| 49 | for i in range(1, N + 1): |
| 50 | if f"({i})" not in s: |
| 51 | return False |
| 52 | return True |
| 53 | |
| 54 | original_str = pred_str |
| 55 | pred_str = pred_str.split("### Question")[0] |
| 56 | pred_answer = pred_str |
| 57 | is_correct = False |
| 58 | correct_ratio = 0 |
| 59 | if reformat_gold_conditions is None: |
| 60 | gold = ans.replace(" and ", "").replace(".", "") |
| 61 | gold_conditions = gold.split(",") |
| 62 | reformat_gold_conditions = [] |
| 63 | for condition in gold_conditions: |
| 64 | gold_condition = condition.strip() # Remove leading and trailing spaces |
| 65 | reformat_gold_conditions.append(gold_condition) |
| 66 | |
| 67 | wrong_reason = "no_conclusion_matched" |
| 68 | for pattern in conclusion_patterns: |
| 69 | pred = pred_str.split(pattern) |
| 70 | if len(pred) > 1: |
| 71 | if len(pred[1]) > 0: # if the matched the answer is not empty |
no test coverage detected