(self,
generations,
samples,
idx=None,
debug=False)
| 191 | return result[0] |
| 192 | |
| 193 | def evaluate_generations(self, |
| 194 | generations, |
| 195 | samples, |
| 196 | idx=None, |
| 197 | debug=False): |
| 198 | assert len(generations.keys()) == len(samples) |
| 199 | results = {} |
| 200 | idx = 0 |
| 201 | for task_id, problem_generations in generations.items(): |
| 202 | sample = samples.iloc[idx] |
| 203 | res = [] |
| 204 | # loop over the generations |
| 205 | for o_idx, o in enumerate(problem_generations): |
| 206 | curr_res = [-2] |
| 207 | try: |
| 208 | curr_res = self.check_correctness(sample, |
| 209 | o, |
| 210 | timeout=TIMEOUT, |
| 211 | debug=debug) |
| 212 | if debug: |
| 213 | print(f'\nSuccessful compilation of task {o_idx}!') |
| 214 | fixed = [] |
| 215 | for e in curr_res: |
| 216 | if isinstance(e, np.ndarray): |
| 217 | e = e.item(0) |
| 218 | if isinstance(e, np.bool_): |
| 219 | e = bool(e) |
| 220 | fixed.append(e) |
| 221 | curr_res = fixed |
| 222 | if not np.all(curr_res): |
| 223 | if debug: |
| 224 | print('Results were not True for all test cases') |
| 225 | except Exception as e: |
| 226 | if debug: |
| 227 | print( |
| 228 | f'Compilation failed, test framework exception = {repr(e)}{e}\n' # noqa: E501 |
| 229 | ) |
| 230 | break |
| 231 | finally: |
| 232 | assert isinstance(curr_res, list) |
| 233 | res.append(curr_res) |
| 234 | results[task_id] = res |
| 235 | idx += 1 |
| 236 | return results |
| 237 | |
| 238 | def estimate_pass_at_k(self, num_samples, num_correct, k): |
| 239 | """Estimates pass@k of each problem and returns them in an array.""" |
no test coverage detected