Skip to content

Commit 84ae313

Browse files
Merge pull request #616 from UiPath/feature/eval-trajectory
feat: trajectory eval
2 parents 2b421c1 + c72bd50 commit 84ae313

6 files changed

Lines changed: 269 additions & 13 deletions

File tree

‎pyproject.toml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
11
[project]
22
name = "uipath"
3-
version = "2.1.66"
3+
version = "2.1.67"
44
description = "Python SDK and CLI for UiPath Platform, enabling programmatic interaction with automation services, process management, and deployment tools."
55
readme = { file = "README.md", content-type = "text/markdown" }
66
requires-python = ">=3.10"

‎src/uipath/_cli/_evals/_evaluator_factory.py‎

Lines changed: 18 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -123,6 +123,22 @@ def _create_llm_as_judge_evaluator(
123123
@staticmethod
124124
def _create_trajectory_evaluator(
125125
base_params: EvaluatorBaseParams, data: Dict[str, Any]
126-
) -> TrajectoryEvaluator[Any]:
126+
) -> TrajectoryEvaluator:
127127
"""Create a trajectory evaluator."""
128-
raise NotImplementedError()
128+
prompt = data.get("prompt", "")
129+
if not prompt:
130+
raise ValueError("Trajectory evaluator must include 'prompt' field")
131+
132+
model = data.get("model", "")
133+
if not model:
134+
raise ValueError("LLM evaluator must include 'model' field")
135+
if model == "same-as-agent":
136+
raise ValueError(
137+
"'same-as-agent' model option is not supported by coded agents evaluations. Please select a specific model for the evaluator."
138+
)
139+
140+
return TrajectoryEvaluator(
141+
**base_params.model_dump(),
142+
prompt=prompt,
143+
model=model,
144+
)
Lines changed: 115 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,115 @@
1+
"""Trajectory evaluation span model for serializing span data in evaluations."""
2+
3+
from dataclasses import dataclass
4+
from typing import Any, Dict, List, Optional
5+
6+
from opentelemetry.sdk.trace import ReadableSpan
7+
from pydantic import BaseModel
8+
9+
10+
@dataclass
11+
class TrajectoryEvaluationSpan:
12+
"""Simplified span representation for trajectory evaluation.
13+
14+
Contains span information needed for evaluating agent execution paths,
15+
excluding timestamps which are not useful for trajectory analysis.
16+
"""
17+
18+
name: str
19+
status: str
20+
attributes: Dict[str, Any]
21+
parent_name: Optional[str] = None
22+
events: Optional[List[Dict[str, Any]]] = None
23+
24+
def __post_init__(self):
25+
"""Initialize default values."""
26+
if self.events is None:
27+
self.events = []
28+
29+
@classmethod
30+
def from_readable_span(
31+
cls, span: ReadableSpan, parent_spans: Optional[Dict[int, str]] = None
32+
) -> "TrajectoryEvaluationSpan":
33+
"""Convert a ReadableSpan to a TrajectoryEvaluationSpan.
34+
35+
Args:
36+
span: The OpenTelemetry ReadableSpan to convert
37+
parent_spans: Optional mapping of span IDs to names for parent lookup
38+
39+
Returns:
40+
TrajectoryEvaluationSpan with relevant data extracted
41+
"""
42+
# Extract status
43+
status_map = {0: "unset", 1: "ok", 2: "error"}
44+
status = status_map.get(span.status.status_code.value, "unknown")
45+
46+
# Extract attributes - keep all attributes for now
47+
attributes = {}
48+
if span.attributes:
49+
attributes = dict(span.attributes)
50+
51+
# Get parent name if available
52+
parent_name = None
53+
if span.parent and parent_spans and span.parent.span_id in parent_spans:
54+
parent_name = parent_spans[span.parent.span_id]
55+
56+
# Extract events (without timestamps)
57+
events = []
58+
if hasattr(span, "events") and span.events:
59+
for event in span.events:
60+
event_data = {
61+
"name": event.name,
62+
"attributes": dict(event.attributes) if event.attributes else {},
63+
}
64+
events.append(event_data)
65+
66+
return cls(
67+
name=span.name,
68+
status=status,
69+
attributes=attributes,
70+
parent_name=parent_name,
71+
events=events,
72+
)
73+
74+
def to_dict(self) -> Dict[str, Any]:
75+
"""Convert to dictionary for JSON serialization."""
76+
return {
77+
"name": self.name,
78+
"status": self.status,
79+
"parent_name": self.parent_name,
80+
"attributes": self.attributes,
81+
"events": self.events,
82+
}
83+
84+
85+
class TrajectoryEvaluationTrace(BaseModel):
86+
"""Container for a collection of trajectory evaluation spans."""
87+
88+
spans: List[TrajectoryEvaluationSpan]
89+
90+
@classmethod
91+
def from_readable_spans(
92+
cls, spans: List[ReadableSpan]
93+
) -> "TrajectoryEvaluationTrace":
94+
"""Convert a list of ReadableSpans to TrajectoryEvaluationTrace.
95+
96+
Args:
97+
spans: List of OpenTelemetry ReadableSpans to convert
98+
99+
Returns:
100+
TrajectoryEvaluationTrace with converted spans
101+
"""
102+
# Create a mapping of span IDs to names for parent lookup
103+
span_id_to_name = {span.get_span_context().span_id: span.name for span in spans}
104+
105+
evaluation_spans = [
106+
TrajectoryEvaluationSpan.from_readable_span(span, span_id_to_name)
107+
for span in spans
108+
]
109+
110+
return cls(spans=evaluation_spans)
111+
112+
class Config:
113+
"""Pydantic configuration."""
114+
115+
arbitrary_types_allowed = True

‎src/uipath/_cli/_evals/_runtime.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -264,6 +264,7 @@ async def run_evaluator(
264264
agent_input=eval_item.inputs,
265265
agent_output=execution_output.result.output or {},
266266
agent_trace=execution_output.spans,
267+
expected_agent_behavior=eval_item.expected_agent_behavior,
267268
)
268269

269270
result = await evaluator.evaluate(
Lines changed: 133 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -1,36 +1,159 @@
11
"""Trajectory evaluator for analyzing execution paths and decision sequences."""
22

3-
from typing import TypeVar
3+
import json
4+
from typing import Any, Optional
45

6+
from opentelemetry.sdk.trace import ReadableSpan
7+
from pydantic import field_validator
8+
9+
from uipath._cli._evals._models._trajectory_span import TrajectoryEvaluationTrace
510
from uipath.eval.models import EvaluationResult
611

7-
from ..models.models import AgentExecution
12+
from ..._services import UiPathLlmChatService
13+
from ..._utils.constants import COMMUNITY_agents_SUFFIX
14+
from ..models.models import AgentExecution, LLMResponse, NumericEvaluationResult
815
from .base_evaluator import BaseEvaluator
916

10-
T = TypeVar("T")
11-
1217

13-
class TrajectoryEvaluator(BaseEvaluator[T]):
18+
class TrajectoryEvaluator(BaseEvaluator[dict[str, Any]]):
1419
"""Evaluator that analyzes the trajectory/path taken to reach outputs."""
1520

21+
prompt: str
22+
model: str
23+
expected_agent_behavior_placeholder: str = "{{ExpectedAgentBehavior}}"
24+
agent_run_history_placeholder: str = "{{AgentRunHistory}}"
25+
llm: Optional[UiPathLlmChatService] = None
26+
27+
@field_validator("prompt")
28+
@classmethod
29+
def validate_prompt_placeholder(cls, v: str) -> str:
30+
"""Validate that prompt contains required placeholders."""
31+
if "{{ExpectedAgentBehavior}}" not in v or "{{AgentRunHistory}}" not in v:
32+
raise ValueError(
33+
"Prompt must contain {ExpectedAgentBehavior} and {{AgentRunHistory}} placeholders"
34+
)
35+
return v
36+
37+
def model_post_init(self, __context):
38+
"""Initialize the LLM service after model creation."""
39+
super().model_post_init(__context)
40+
self._initialize_llm()
41+
42+
def _initialize_llm(self):
43+
"""Initialize the LLM used for evaluation."""
44+
from uipath import UiPath
45+
46+
uipath = UiPath()
47+
self.llm = uipath.llm
48+
1649
async def evaluate(
17-
self, agent_execution: AgentExecution, evaluation_criteria: T
50+
self,
51+
agent_execution: AgentExecution,
52+
evaluation_criteria: dict[str, Any],
1853
) -> EvaluationResult:
1954
"""Evaluate using trajectory analysis.
2055
21-
Analyzes the execution path and decision sequence taken by the agent
22-
to assess the quality of the reasoning process.
56+
Analyzes the execution path and decision sequence taken by the agent.
2357
2458
Args:
2559
agent_execution: The execution details containing:
2660
- agent_input: The input received by the agent
2761
- actual_output: The actual output from the agent
28-
- spans: The execution spans to use for the evaluation
62+
- agent_trace: The execution spans to use for the evaluation
63+
- expected_agent_behavior: The expected agent behavior
2964
evaluation_criteria: The criteria to evaluate
3065
Returns:
3166
EvaluationResult: Score based on trajectory analysis
3267
3368
Raises:
3469
NotImplementedError: This evaluator is not yet implemented
3570
"""
36-
raise NotImplementedError()
71+
evaluation_prompt = self._create_evaluation_prompt(
72+
expected_agent_behavior=agent_execution.expected_agent_behavior,
73+
agent_run_history=agent_execution.agent_trace,
74+
)
75+
76+
llm_response = await self._get_llm_response(evaluation_prompt)
77+
78+
return NumericEvaluationResult(
79+
score=llm_response.score,
80+
details=llm_response.justification,
81+
)
82+
83+
def _create_evaluation_prompt(
84+
self,
85+
expected_agent_behavior: Any,
86+
agent_run_history: Any,
87+
) -> str:
88+
"""Create the evaluation prompt for the LLM."""
89+
formatted_prompt = self.prompt.replace(
90+
self.expected_agent_behavior_placeholder,
91+
str(expected_agent_behavior),
92+
)
93+
94+
# Trim extra properties from the spans (such as timestamps which are not relevant to the eval)
95+
if (
96+
isinstance(agent_run_history, list)
97+
and agent_run_history
98+
and isinstance(agent_run_history[0], ReadableSpan)
99+
):
100+
trajectory_trace = TrajectoryEvaluationTrace.from_readable_spans(
101+
agent_run_history
102+
)
103+
agent_run_history = str(trajectory_trace.spans)
104+
else:
105+
agent_run_history = str(agent_run_history)
106+
107+
formatted_prompt = formatted_prompt.replace(
108+
self.agent_run_history_placeholder,
109+
agent_run_history,
110+
)
111+
112+
return formatted_prompt
113+
114+
async def _get_llm_response(self, evaluation_prompt: str) -> LLMResponse:
115+
"""Get response from the LLM.
116+
117+
Args:
118+
evaluation_prompt: The formatted prompt to send to the LLM
119+
120+
Returns:
121+
LLMResponse with score and justification
122+
"""
123+
if not self.llm:
124+
raise ValueError("LLM service not initialized")
125+
126+
model = self.model
127+
if model.endswith(COMMUNITY_agents_SUFFIX):
128+
model = model.replace(COMMUNITY_agents_SUFFIX, "")
129+
130+
# Prepare the request
131+
request_data = {
132+
"model": model,
133+
"messages": [{"role": "user", "content": evaluation_prompt}],
134+
"response_format": {
135+
"type": "json_schema",
136+
"json_schema": {
137+
"name": "evaluation_response",
138+
"schema": {
139+
"type": "object",
140+
"properties": {
141+
"score": {
142+
"type": "number",
143+
"minimum": 0,
144+
"maximum": 100,
145+
"description": "Score between 0 and 100",
146+
},
147+
"justification": {
148+
"type": "string",
149+
"description": "Explanation for the score",
150+
},
151+
},
152+
"required": ["score", "justification"],
153+
},
154+
},
155+
},
156+
}
157+
158+
response = await self.llm.chat_completions(**request_data)
159+
return LLMResponse(**json.loads(response.choices[-1].message.content))

‎src/uipath/eval/models/models.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -15,6 +15,7 @@ class AgentExecution(BaseModel):
1515
agent_input: Optional[Dict[str, Any]]
1616
agent_output: Dict[str, Any]
1717
agent_trace: list[ReadableSpan]
18+
expected_agent_behavior: Optional[str] = None
1819

1920

2021
class LLMResponse(BaseModel):

0 commit comments

Comments
 (0)