(question, traj)
| 344 | return False |
| 345 | |
| 346 | def _prepare_prompt(question, traj): |
| 347 | if not traj: traj = "" |
| 348 | |
| 349 | |
| 350 | clean_traj = traj.replace("<<<EVAL_RESULT: SUCCESS>>>", "") \ |
| 351 | .replace("<<<EVAL_RESULT: FAILURE>>>", "") \ |
| 352 | .strip() |
| 353 | |
| 354 | |
| 355 | MAX_CHARS = 25000 |
| 356 | if len(clean_traj) > MAX_CHARS: |
| 357 | half = MAX_CHARS // 2 |
| 358 | clean_traj = f"{clean_traj[:half]}\n\n... [TRUNCATED] ...\n\n{clean_traj[-half:]}" |
| 359 | |
| 360 | |
| 361 | system_message = "You are a logic consistency evaluator." |
| 362 | user_message = APPWORLD_JUDGE_PROMPT.format(question=question, trajectory=clean_traj) |
| 363 | |
| 364 | return f"{system_message}\n\n{user_message}" |
| 365 | |
| 366 | def appworld_reward_fn(question, answer1, answer2, standard_answer): |
| 367 | SUCCESS_TAG = "<<<EVAL_RESULT: SUCCESS>>>" |
no outgoing calls
no test coverage detected