|
1 | 1 | """Trajectory evaluator for analyzing execution paths and decision sequences.""" |
2 | 2 |
|
3 | | -from typing import TypeVar |
| 3 | +import json |
| 4 | +from typing import Any, Optional |
4 | 5 |
|
| 6 | +from opentelemetry.sdk.trace import ReadableSpan |
| 7 | +from pydantic import field_validator |
| 8 | + |
| 9 | +from uipath._cli._evals._models._trajectory_span import TrajectoryEvaluationTrace |
5 | 10 | from uipath.eval.models import EvaluationResult |
6 | 11 |
|
7 | | -from ..models.models import AgentExecution |
| 12 | +from ..._services import UiPathLlmChatService |
| 13 | +from ..._utils.constants import COMMUNITY_agents_SUFFIX |
| 14 | +from ..models.models import AgentExecution, LLMResponse, NumericEvaluationResult |
8 | 15 | from .base_evaluator import BaseEvaluator |
9 | 16 |
|
10 | | -T = TypeVar("T") |
11 | | - |
12 | 17 |
|
13 | | -class TrajectoryEvaluator(BaseEvaluator[T]): |
| 18 | +class TrajectoryEvaluator(BaseEvaluator[dict[str, Any]]): |
14 | 19 | """Evaluator that analyzes the trajectory/path taken to reach outputs.""" |
15 | 20 |
|
| 21 | + prompt: str |
| 22 | + model: str |
| 23 | + expected_agent_behavior_placeholder: str = "{{ExpectedAgentBehavior}}" |
| 24 | + agent_run_history_placeholder: str = "{{AgentRunHistory}}" |
| 25 | + llm: Optional[UiPathLlmChatService] = None |
| 26 | + |
| 27 | + @field_validator("prompt") |
| 28 | + @classmethod |
| 29 | + def validate_prompt_placeholder(cls, v: str) -> str: |
| 30 | + """Validate that prompt contains required placeholders.""" |
| 31 | + if "{{ExpectedAgentBehavior}}" not in v or "{{AgentRunHistory}}" not in v: |
| 32 | + raise ValueError( |
| 33 | + "Prompt must contain {ExpectedAgentBehavior} and {{AgentRunHistory}} placeholders" |
| 34 | + ) |
| 35 | + return v |
| 36 | + |
| 37 | + def model_post_init(self, __context): |
| 38 | + """Initialize the LLM service after model creation.""" |
| 39 | + super().model_post_init(__context) |
| 40 | + self._initialize_llm() |
| 41 | + |
| 42 | + def _initialize_llm(self): |
| 43 | + """Initialize the LLM used for evaluation.""" |
| 44 | + from uipath import UiPath |
| 45 | + |
| 46 | + uipath = UiPath() |
| 47 | + self.llm = uipath.llm |
| 48 | + |
16 | 49 | async def evaluate( |
17 | | - self, agent_execution: AgentExecution, evaluation_criteria: T |
| 50 | + self, |
| 51 | + agent_execution: AgentExecution, |
| 52 | + evaluation_criteria: dict[str, Any], |
18 | 53 | ) -> EvaluationResult: |
19 | 54 | """Evaluate using trajectory analysis. |
20 | 55 |
|
21 | | - Analyzes the execution path and decision sequence taken by the agent |
22 | | - to assess the quality of the reasoning process. |
| 56 | + Analyzes the execution path and decision sequence taken by the agent. |
23 | 57 |
|
24 | 58 | Args: |
25 | 59 | agent_execution: The execution details containing: |
26 | 60 | - agent_input: The input received by the agent |
27 | 61 | - actual_output: The actual output from the agent |
28 | | - - spans: The execution spans to use for the evaluation |
| 62 | + - agent_trace: The execution spans to use for the evaluation |
| 63 | + - expected_agent_behavior: The expected agent behavior |
29 | 64 | evaluation_criteria: The criteria to evaluate |
30 | 65 | Returns: |
31 | 66 | EvaluationResult: Score based on trajectory analysis |
32 | 67 |
|
33 | 68 | Raises: |
34 | 69 | NotImplementedError: This evaluator is not yet implemented |
35 | 70 | """ |
36 | | - raise NotImplementedError() |
| 71 | + evaluation_prompt = self._create_evaluation_prompt( |
| 72 | + expected_agent_behavior=agent_execution.expected_agent_behavior, |
| 73 | + agent_run_history=agent_execution.agent_trace, |
| 74 | + ) |
| 75 | + |
| 76 | + llm_response = await self._get_llm_response(evaluation_prompt) |
| 77 | + |
| 78 | + return NumericEvaluationResult( |
| 79 | + score=llm_response.score, |
| 80 | + details=llm_response.justification, |
| 81 | + ) |
| 82 | + |
| 83 | + def _create_evaluation_prompt( |
| 84 | + self, |
| 85 | + expected_agent_behavior: Any, |
| 86 | + agent_run_history: Any, |
| 87 | + ) -> str: |
| 88 | + """Create the evaluation prompt for the LLM.""" |
| 89 | + formatted_prompt = self.prompt.replace( |
| 90 | + self.expected_agent_behavior_placeholder, |
| 91 | + str(expected_agent_behavior), |
| 92 | + ) |
| 93 | + |
| 94 | + # Trim extra properties from the spans (such as timestamps which are not relevant to the eval) |
| 95 | + if ( |
| 96 | + isinstance(agent_run_history, list) |
| 97 | + and agent_run_history |
| 98 | + and isinstance(agent_run_history[0], ReadableSpan) |
| 99 | + ): |
| 100 | + trajectory_trace = TrajectoryEvaluationTrace.from_readable_spans( |
| 101 | + agent_run_history |
| 102 | + ) |
| 103 | + agent_run_history = str(trajectory_trace.spans) |
| 104 | + else: |
| 105 | + agent_run_history = str(agent_run_history) |
| 106 | + |
| 107 | + formatted_prompt = formatted_prompt.replace( |
| 108 | + self.agent_run_history_placeholder, |
| 109 | + agent_run_history, |
| 110 | + ) |
| 111 | + |
| 112 | + return formatted_prompt |
| 113 | + |
| 114 | + async def _get_llm_response(self, evaluation_prompt: str) -> LLMResponse: |
| 115 | + """Get response from the LLM. |
| 116 | +
|
| 117 | + Args: |
| 118 | + evaluation_prompt: The formatted prompt to send to the LLM |
| 119 | +
|
| 120 | + Returns: |
| 121 | + LLMResponse with score and justification |
| 122 | + """ |
| 123 | + if not self.llm: |
| 124 | + raise ValueError("LLM service not initialized") |
| 125 | + |
| 126 | + model = self.model |
| 127 | + if model.endswith(COMMUNITY_agents_SUFFIX): |
| 128 | + model = model.replace(COMMUNITY_agents_SUFFIX, "") |
| 129 | + |
| 130 | + # Prepare the request |
| 131 | + request_data = { |
| 132 | + "model": model, |
| 133 | + "messages": [{"role": "user", "content": evaluation_prompt}], |
| 134 | + "response_format": { |
| 135 | + "type": "json_schema", |
| 136 | + "json_schema": { |
| 137 | + "name": "evaluation_response", |
| 138 | + "schema": { |
| 139 | + "type": "object", |
| 140 | + "properties": { |
| 141 | + "score": { |
| 142 | + "type": "number", |
| 143 | + "minimum": 0, |
| 144 | + "maximum": 100, |
| 145 | + "description": "Score between 0 and 100", |
| 146 | + }, |
| 147 | + "justification": { |
| 148 | + "type": "string", |
| 149 | + "description": "Explanation for the score", |
| 150 | + }, |
| 151 | + }, |
| 152 | + "required": ["score", "justification"], |
| 153 | + }, |
| 154 | + }, |
| 155 | + }, |
| 156 | + } |
| 157 | + |
| 158 | + response = await self.llm.chat_completions(**request_data) |
| 159 | + return LLMResponse(**json.loads(response.choices[-1].message.content)) |
0 commit comments