Skip to content

Commit cfdc86d

Browse files
authored
Merge pull request #787 from UiPath/akshaya/runtime_details
feat(ReturnAgentExecution): return agent execution details as part of evaluation output
2 parents 1a60bec + 5ae7f21 commit cfdc86d

7 files changed

Lines changed: 59 additions & 24 deletions

File tree

‎pyproject.toml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
11
[project]
22
name = "uipath"
3-
version = "2.1.112"
3+
version = "2.1.113"
44
description = "Python SDK and CLI for UiPath Platform, enabling programmatic interaction with automation services, process management, and deployment tools."
55
readme = { file = "README.md", content-type = "text/markdown" }
66
requires-python = ">=3.10"

‎samples/calculator/evals/eval-sets/legacy.json‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -21,7 +21,7 @@
2121
"expectedOutput": {
2222
"result": 2
2323
},
24-
"expectedAgentBehavior": "",
24+
"expectedAgentBehavior": "The operation should produce the right output.",
2525
"evalSetId": "default-eval-set-id",
2626
"createdAt": "2025-09-04T18:54:58.378Z",
2727
"updatedAt": "2025-09-04T18:55:55.416Z"
@@ -37,7 +37,7 @@
3737
"expectedOutput": {
3838
"result": 2
3939
},
40-
"expectedAgentBehavior": "",
40+
"expectedAgentBehavior": "The operation should produce the right output.",
4141
"mockingStrategy": {
4242
"type": "mockito",
4343
"behaviors": [
@@ -73,7 +73,7 @@
7373
"expectedOutput": {
7474
"result": 2
7575
},
76-
"expectedAgentBehavior": "",
76+
"expectedAgentBehavior": "The operation should produce the right output.",
7777
"mockingStrategy": {
7878
"type": "llm",
7979
"prompt": "The random operator is '+'.",
@@ -98,7 +98,7 @@
9898
"expectedOutput": {
9999
"result": 35
100100
},
101-
"expectedAgentBehavior": "",
101+
"expectedAgentBehavior": "The operation should produce the right output.",
102102
"inputMockingStrategy": {
103103
"prompt": "Generate a multiplication calculation where the first number is 5 and the second number is 7",
104104
"model": {

‎samples/calculator/evals/evaluators/legacy-trajectory.json‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,7 +6,7 @@
66
"category": 3,
77
"type": 7,
88
"prompt": "Evaluate the agent's execution trajectory based on the expected behavior.\n\nExpected Agent Behavior: {{ExpectedAgentBehavior}}\nAgent Run History: {{AgentRunHistory}}\n\nProvide a score from 0-100 based on how well the agent followed the expected trajectory.",
9-
"model": "gpt-4o-mini",
9+
"model": "gpt-4.1-2025-04-14",
1010
"targetOutputKey": "*",
1111
"createdAt": "2025-06-26T17:45:39.651Z",
1212
"updatedAt": "2025-06-26T17:45:39.651Z"

‎src/uipath/_cli/_evals/_models/_output.py‎

Lines changed: 22 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -8,7 +8,11 @@
88
from pydantic_core import core_schema
99

1010
from uipath._cli._runtime._contracts import UiPathRuntimeResult
11-
from uipath.eval.models.models import EvaluationResult, ScoreType
11+
from uipath.eval.models.models import (
12+
EvaluationResult,
13+
ScoreType,
14+
TrajectoryEvaluationTrace,
15+
)
1216

1317

1418
class UiPathEvalRunExecutionOutput(BaseModel):
@@ -22,6 +26,22 @@ class UiPathEvalRunExecutionOutput(BaseModel):
2226
result: UiPathRuntimeResult
2327

2428

29+
class UiPathSerializableEvalRunExecutionOutput(BaseModel):
30+
execution_time: float
31+
trace: TrajectoryEvaluationTrace
32+
result: UiPathRuntimeResult
33+
34+
35+
def convert_eval_execution_output_to_serializable(
36+
output: UiPathEvalRunExecutionOutput,
37+
) -> UiPathSerializableEvalRunExecutionOutput:
38+
return UiPathSerializableEvalRunExecutionOutput(
39+
execution_time=output.execution_time,
40+
result=output.result,
41+
trace=TrajectoryEvaluationTrace.from_readable_spans(output.spans),
42+
)
43+
44+
2545
class EvaluationResultDto(BaseModel):
2646
model_config = ConfigDict(alias_generator=to_camel, populate_by_name=True)
2747

@@ -67,19 +87,13 @@ class EvaluationRunResultDto(BaseModel):
6787
evaluator_id: str
6888
result: EvaluationResultDto
6989

70-
@model_serializer(mode="wrap")
71-
def serialize_model(self, serializer, info):
72-
data = serializer(self)
73-
if isinstance(data, dict):
74-
data.pop("evaluatorId", None)
75-
return data
76-
7790

7891
class EvaluationRunResult(BaseModel):
7992
model_config = ConfigDict(alias_generator=to_camel, populate_by_name=True)
8093

8194
evaluation_name: str
8295
evaluation_run_results: List[EvaluationRunResultDto]
96+
agent_execution_output: Optional[UiPathSerializableEvalRunExecutionOutput] = None
8397

8498
@property
8599
def score(self) -> float:

‎src/uipath/_cli/_evals/_runtime.py‎

Lines changed: 21 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -38,6 +38,7 @@
3838
)
3939
from .._runtime._logging import ExecutionLogHandler
4040
from .._utils._eval_set import EvalHelpers
41+
from ..models.runtime_schema import Entrypoint
4142
from ._evaluator_factory import EvaluatorFactory
4243
from ._models._evaluation_set import (
4344
AnyEvaluationItem,
@@ -53,6 +54,7 @@
5354
EvaluationRunResultDto,
5455
UiPathEvalOutput,
5556
UiPathEvalRunExecutionOutput,
57+
convert_eval_execution_output_to_serializable,
5658
)
5759
from ._span_collection import ExecutionSpanCollector
5860
from .mocks.mocks import (
@@ -147,6 +149,7 @@ class UiPathEvalContext(UiPathRuntimeContext):
147149
workers: Optional[int] = 1
148150
eval_set: Optional[str] = None
149151
eval_ids: Optional[List[str]] = None
152+
verbose: bool = False
150153

151154

152155
class UiPathEvalRuntime(UiPathBaseRuntime, Generic[T, C]):
@@ -173,6 +176,15 @@ def __init__(
173176

174177
self.logs_exporter: ExecutionLogsExporter = ExecutionLogsExporter()
175178
self.execution_id = str(uuid.uuid4())
179+
self.entrypoint: Optional[Entrypoint] = None
180+
181+
async def get_entrypoint(self):
182+
if not self.entrypoint:
183+
temp_runtime = self.factory.new_runtime(
184+
entrypoint=self.context.entrypoint, runtime_dir=os.getcwd()
185+
)
186+
self.entrypoint = await temp_runtime.get_entrypoint()
187+
return self.entrypoint
176188

177189
@classmethod
178190
def from_eval_context(
@@ -187,13 +199,6 @@ async def execute(self) -> UiPathRuntimeResult:
187199
if self.context.eval_set is None:
188200
raise ValueError("eval_set must be provided for evaluation runs")
189201

190-
# Get entrypoint from a temporary runtime
191-
temp_context = self.factory.new_context(
192-
entrypoint=self.context.entrypoint, runtime_dir=os.getcwd()
193-
)
194-
temp_runtime = self.factory.from_context(temp_context)
195-
self.entrypoint = await temp_runtime.get_entrypoint()
196-
197202
event_bus = self.event_bus
198203

199204
# Load eval set (path is already resolved in cli_eval.py)
@@ -360,6 +365,12 @@ async def _execute_eval(
360365

361366
try:
362367
agent_execution_output = await self.execute_runtime(eval_item, execution_id)
368+
if self.context.verbose:
369+
evaluation_run_results.agent_execution_output = (
370+
convert_eval_execution_output_to_serializable(
371+
agent_execution_output
372+
)
373+
)
363374
evaluation_item_results: list[EvalItemResult] = []
364375

365376
for evaluator in evaluators:
@@ -477,7 +488,9 @@ async def _generate_input_for_eval(
477488
self, eval_item: AnyEvaluationItem
478489
) -> AnyEvaluationItem:
479490
"""Use LLM to generate a mock input for an evaluation item."""
480-
generated_input = await generate_llm_input(eval_item, self.entrypoint.input)
491+
generated_input = await generate_llm_input(
492+
eval_item, (await self.get_entrypoint()).input
493+
)
481494
updated_eval_item = eval_item.model_copy(update={"inputs": generated_input})
482495
return updated_eval_item
483496

‎tests/cli/eval/test_evaluate.py‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -2,6 +2,7 @@
22
from typing import Any
33

44
from uipath._cli._evals._evaluate import evaluate
5+
from uipath._cli._evals._models._output import UiPathEvalOutput
56
from uipath._cli._evals._runtime import UiPathEvalContext
67
from uipath._cli._runtime._contracts import UiPathRuntimeContext, UiPathRuntimeFactory
78
from uipath._cli._runtime._runtime import UiPathRuntime
@@ -42,11 +43,18 @@ def __init__(self):
4243
# Act
4344
result = await evaluate(MyFactory(), context, event_bus)
4445

45-
# Assert
46+
# Assert that the output is json-serializable
47+
UiPathEvalOutput.model_validate(result.output).model_dump_json()
4648
assert result.output
4749
assert (
4850
result.output["evaluationSetResults"][0]["evaluationRunResults"][0]["result"][
4951
"score"
5052
]
5153
== 100.0
5254
)
55+
assert (
56+
result.output["evaluationSetResults"][0]["evaluationRunResults"][0][
57+
"evaluatorId"
58+
]
59+
== "equality"
60+
)

‎uv.lock‎

Lines changed: 1 addition & 1 deletion
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.

0 commit comments

Comments
 (0)