MCPcopy Create free account
hub / github.com/bruno686/VisPlay / compute_score

Function compute_score

validation_examples/reward_function/validation.py:156–187  ·  view source on GitHub ↗
(predicts: List[str], ground_truths: List[str], questions: List[str], description_answers: List[str], format_weight: float = 0.1, images: Optional[List[object]] = None)

Source from the content-addressed store, hash-verified

154
155
156def compute_score(predicts: List[str], ground_truths: List[str], questions: List[str], description_answers: List[str], format_weight: float = 0.1, images: Optional[List[object]] = None) -> List[Dict[str, float]]:
157 scores = []
158 results = []
159 for predict, ground_truth in zip(predicts, ground_truths):
160 predict = re.sub(r"\s*(<|>|/)\s*", r"\1", predict) # handle qwen2.5vl-32b format
161 # format_score = format_reward(predict)
162 # accuracy_score = accuracy_reward(predict, ground_truth)
163 dirty_results = match(predict)
164 questions = dirty_results['question']
165 types = dirty_results['type']
166 answers = dirty_results['answer']
167
168
169 if questions and answers and types:
170 try:
171 question = questions[-1].strip()
172 answer = answers[-1].strip()
173 results.append({"question": question, "answer": answer, "types": types})
174 except:
175 results.append({"question": "", "answer": "", "types": ""})
176 else:
177 results.append({"question": "", "answer": "", "types": ""})
178
179 final_results = generate_results(results)
180 penalty = cluster_share_per_problem([result['question'] for result in final_results], distance_threshold=0.5)
181
182 assert len(penalty) == len(final_results)
183 scores = []
184 for i in range(len(final_results)):
185 final_score = (min(final_results[i]["score"],1-final_results[i]["score"]) if final_results[i]['question'] else -1)-penalty[i]
186 scores.append({"overall": final_score,"format": 1 if final_results[i]['question'] else 0,"accuracy": penalty[i]})
187 return scores
188

Callers

nothing calls this directly

Calls 3

matchFunction · 0.70
generate_resultsFunction · 0.70

Tested by

no test coverage detected