(predicts: List[str], ground_truths: List[str], questions: List[str], description_answers: List[str], format_weight: float = 0.1, images: Optional[List[object]] = None)
| 154 | |
| 155 | |
| 156 | def compute_score(predicts: List[str], ground_truths: List[str], questions: List[str], description_answers: List[str], format_weight: float = 0.1, images: Optional[List[object]] = None) -> List[Dict[str, float]]: |
| 157 | scores = [] |
| 158 | results = [] |
| 159 | for predict, ground_truth in zip(predicts, ground_truths): |
| 160 | predict = re.sub(r"\s*(<|>|/)\s*", r"\1", predict) # handle qwen2.5vl-32b format |
| 161 | # format_score = format_reward(predict) |
| 162 | # accuracy_score = accuracy_reward(predict, ground_truth) |
| 163 | dirty_results = match(predict) |
| 164 | questions = dirty_results['question'] |
| 165 | types = dirty_results['type'] |
| 166 | answers = dirty_results['answer'] |
| 167 | |
| 168 | |
| 169 | if questions and answers and types: |
| 170 | try: |
| 171 | question = questions[-1].strip() |
| 172 | answer = answers[-1].strip() |
| 173 | results.append({"question": question, "answer": answer, "types": types}) |
| 174 | except: |
| 175 | results.append({"question": "", "answer": "", "types": ""}) |
| 176 | else: |
| 177 | results.append({"question": "", "answer": "", "types": ""}) |
| 178 | |
| 179 | final_results = generate_results(results) |
| 180 | penalty = cluster_share_per_problem([result['question'] for result in final_results], distance_threshold=0.5) |
| 181 | |
| 182 | assert len(penalty) == len(final_results) |
| 183 | scores = [] |
| 184 | for i in range(len(final_results)): |
| 185 | final_score = (min(final_results[i]["score"],1-final_results[i]["score"]) if final_results[i]['question'] else -1)-penalty[i] |
| 186 | scores.append({"overall": final_score,"format": 1 if final_results[i]['question'] else 0,"accuracy": penalty[i]}) |
| 187 | return scores |
| 188 |
nothing calls this directly
no test coverage detected