Evaluate a single tool for the query.
(self, query: str, tool: Tool)
| 553 | await self.handler.state.precheck_step_done(self.STEP_NAME) |
| 554 | |
| 555 | async def _evaluate_tool(self, query: str, tool: Tool) -> dict: |
| 556 | """Evaluate a single tool for the query.""" |
| 557 | if not tool.prompt: |
| 558 | return {"tool": tool, "score": 0, "justification": "No prompt defined"} |
| 559 | |
| 560 | # Fill prompt using the proper mechanism that includes all context |
| 561 | filled_prompt = fill_prompt(tool.prompt, self.handler) |
| 562 | |
| 563 | try: |
| 564 | # Use high level for all tools to ensure fair evaluation timing |
| 565 | level = "high" |
| 566 | start_time = time.time() |
| 567 | response = await ask_llm(filled_prompt, tool.return_structure, level=level, query_params=self.handler.query_params) |
| 568 | end_time = time.time() |
| 569 | end_time - start_time |
| 570 | |
| 571 | result = response or {"score": 0, "justification": "No response from LLM"} |
| 572 | |
| 573 | # Tool evaluation completed |
| 574 | |
| 575 | return {"tool": tool, "result": result, "score": result.get("score", 0)} |
| 576 | except Exception as e: |
| 577 | # Tool evaluation error |
| 578 | logger.error(f"Tool evaluation error for {tool.name}: {e!s}") |
| 579 | return {"tool": tool, "score": 0, "result": {"score": 0, "justification": f"Error: {e!s}"}} |
| 580 | |
| 581 | async def _send_message(self, tool_scores, query, schema_type): |
| 582 | """Send tool selection results as message.""" |
no test coverage detected