Evaluate a batch of question-answer pairs. Args: questions: List of questions answers1: List of first answers answers2: List of second answers standard_answers: List of ground truth answers Returns:
(
self,
questions: List[str],
answers1: List[str],
answers2: List[str],
standard_answers: List[str],
)
| 73 | ) |
| 74 | |
| 75 | def evaluate_batch( |
| 76 | self, |
| 77 | questions: List[str], |
| 78 | answers1: List[str], |
| 79 | answers2: List[str], |
| 80 | standard_answers: List[str], |
| 81 | ) -> Tuple[Dict[str, float], List[float]]: |
| 82 | """ |
| 83 | Evaluate a batch of question-answer pairs. |
| 84 | |
| 85 | Args: |
| 86 | questions: List of questions |
| 87 | answers1: List of first answers |
| 88 | answers2: List of second answers |
| 89 | standard_answers: List of ground truth answers |
| 90 | |
| 91 | Returns: |
| 92 | Tuple of (metrics, individual_results): |
| 93 | - metrics: Dictionary with evaluation metrics: |
| 94 | - win_rate: Proportion of times answers1 wins (0.5 for ties) |
| 95 | - total_comparisons: Total number of comparisons |
| 96 | - wins: Number of times answers1 wins |
| 97 | - losses: Number of times answers2 wins |
| 98 | - ties: Number of ties |
| 99 | - individual_results: List of individual scores for each sample |
| 100 | (1.0 = answers1 wins, 0.0 = answers2 wins, 0.5 = tie) |
| 101 | """ |
| 102 | if not (len(questions) == len(answers1) == len(answers2) == len(standard_answers)): |
| 103 | raise ValueError("All input lists must have the same length") |
| 104 | |
| 105 | wins = 0 |
| 106 | ties = 0 |
| 107 | total = len(questions) |
| 108 | individual_results = [] |
| 109 | |
| 110 | for q, a1, a2, std in zip(questions, answers1, answers2, standard_answers): |
| 111 | result = self.evaluate_single(q, a1, a2, std) |
| 112 | individual_results.append(result) |
| 113 | if result == 1.0: |
| 114 | wins += 1 |
| 115 | elif result == 0.5: |
| 116 | ties += 1 |
| 117 | |
| 118 | losses = (1 - (wins / (total - ties))) if total - ties > 0 else 0.0 |
| 119 | |
| 120 | metrics = { |
| 121 | "win_rate": wins / (total - ties) if total - ties > 0 else 0.0, |
| 122 | "tie_rate": ties / total if total > 0 else 0.0, |
| 123 | "total_comparisons": total, |
| 124 | "wins": wins, |
| 125 | "losses": losses, |
| 126 | "ties": ties, |
| 127 | } |
| 128 | |
| 129 | return metrics, individual_results |
| 130 | |
| 131 | def generate_and_evaluate( |
| 132 | self, |
no test coverage detected