From 762a59c77c1100129a0607e4101fb8249d4f4470 Mon Sep 17 00:00:00 2001 From: Peter Tomko Date: Mon, 5 Oct 2026 22:04:12 +0200 Subject: [PATCH 1/4] feat(gooddata-eval): populate best_run_latency_s for agentic evaluators ItemReport.best_run_latency_s has only ever been set on the single-shot path (_run_one_item times each of its K attempts directly and keeps the best one's own latency). The agentic path drives its K runs and best-of-K selection inside each evaluator, so the runner never had a per-run latency to read -- every agentic kind reported best_run_latency_s: null, which is why tools consuming it (e.g. the Latency Explorer's tool-calls-vs-run-time panel) have always been empty for agentic_*/vis_agentic. Adds AgenticEvalOutcome.best_run_latency_s (and the matching bare annotation on AgenticAssertionError/JudgeResponseError, same pattern already used for timings) and wires it through all 11 agentic evaluators: - general_question, metric_skill, dashboard_skill already track per-run PhaseTimings; best_run_latency_s is derived from the rank-selected run's own timings via the new run_latency_s() helper (excludes langfuse_s, which is deliberately off the critical path). - The other 8 (guardrail, search_tool, alert_skill, anomaly_detection, kda_skill, visualization, what_if, conversation) had no per-run wall-clock at all; each gets a new run_latency_s field on its per-run result, measured the same way _run_one_item already does (time.monotonic() bracketing the whole run). agentic_runner.py copies it onto the ItemReport via a new _apply_best_run_latency(), mirroring the existing _apply_timings(). Verified: full gooddata-eval suite (1382 tests), ruff lint/format, and ty type-check all pass. Added focused tests for the new run_latency_s() helper, _apply_best_run_latency()'s three paths (success/failure/absent), and end-to-end value assertions for general_question and metric_skill (the two kinds with fully deterministic mocked clocks). Co-Authored-By: Claude Sonnet 5 --- .../src/gooddata_eval/cli/agentic_runner.py | 15 ++++++ .../gooddata_eval/core/agentic/alert_skill.py | 8 ++++ .../core/agentic/anomaly_detection.py | 9 ++++ .../core/agentic/conversation.py | 10 ++++ .../core/agentic/dashboard_skill.py | 4 +- .../core/agentic/general_question.py | 5 +- .../gooddata_eval/core/agentic/guardrail.py | 8 ++++ .../gooddata_eval/core/agentic/kda_skill.py | 9 ++++ .../core/agentic/metric_skill.py | 4 +- .../gooddata_eval/core/agentic/search_tool.py | 10 ++++ .../core/agentic/visualization.py | 8 ++++ .../src/gooddata_eval/core/agentic/what_if.py | 9 ++++ .../core/evaluators/_llm_judge.py | 1 + .../src/gooddata_eval/core/models.py | 7 +++ .../src/gooddata_eval/core/timing.py | 10 ++++ .../tests/test_agentic_general_question.py | 4 ++ .../tests/test_agentic_metric_skill.py | 1 + .../tests/test_agentic_runner.py | 48 +++++++++++++++++++ packages/gooddata-eval/tests/test_timing.py | 8 ++++ 19 files changed, 175 insertions(+), 3 deletions(-) diff --git a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py index 048eb6410..08971d7e2 100644 --- a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py @@ -340,6 +340,18 @@ def _apply_timings(item_report: ItemReport, timings: Any) -> None: item_report.simulated_user_latency_s = timings.simulated_user_s +def _apply_best_run_latency(item_report: ItemReport, source: Any) -> None: + """Copy the rank-selected run's own wall time, mirroring the single-shot path's + ``ItemReport.best_run_latency_s`` (see ``core/runner.py``'s ``_run_one_item``). + + Kinds not yet wired to measure it pass None and keep the field at its own None + default rather than reporting an invented number. + """ + best_run_latency_s = getattr(source, "best_run_latency_s", None) + if best_run_latency_s is not None: + item_report.best_run_latency_s = best_run_latency_s + + def run_agentic_items( items: list[DatasetItem], host: str, @@ -425,6 +437,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport: item_report.response_id = response_id item_report.best_detail = detail or {} _apply_timings(item_report, getattr(outcome, "timings", None)) + _apply_best_run_latency(item_report, outcome) _apply_run_counts(item_report, outcome) except AssertionError as exc: item_report.gate_passed = False if gated else None @@ -434,6 +447,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport: item_report.response_id = getattr(exc, "response_id", None) item_report.best_detail = getattr(exc, "detail", None) or {} _apply_timings(item_report, getattr(exc, "timings", None)) + _apply_best_run_latency(item_report, exc) _apply_run_counts(item_report, exc) # Read off the counts, not off the gate: pass^K fails items where runs did pass, # and reporting those as pass_at_k False would contradict the Langfuse score of @@ -448,6 +462,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport: # judge broke should not also report the agent as costing 0s. Kinds that # attach no timings to the exception keep their 0.0 defaults. _apply_timings(item_report, getattr(exc, "timings", None)) + _apply_best_run_latency(item_report, exc) finally: item_report.latency_s = time.perf_counter() - t0 diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py index c1228e89b..bb59fdf45 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py @@ -6,6 +6,7 @@ import json import os import re +import time from dataclasses import dataclass, field from typing import Any @@ -490,6 +491,9 @@ class AlertRunResult: # also report operator/threshold/metric/recipients as False. exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED turns_used: int = 0 + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -670,6 +674,7 @@ def run_agentic_alert_skill( def _run_once(conv_id: str) -> AlertRunResult: alert_id_to_delete: str | None = None + run_started = time.monotonic() try: alert_id: str | None = None actual_args: dict = {} @@ -788,6 +793,7 @@ def _run_once(conv_id: str) -> AlertRunResult: reasoning_step_events=all_reasoning_step_events, exit_reason=exit_reason, turns_used=turns_used, + run_latency_s=time.monotonic() - run_started, ) finally: if alert_id_to_delete: @@ -982,6 +988,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective + exc.best_run_latency_s = best.run_latency_s raise exc return AgenticEvalOutcome( runs_passed=runs_passed, @@ -990,4 +997,5 @@ def _write_scores(ctx: RunTraceContext) -> None: conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py index 778da4fea..daa603064 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py @@ -28,6 +28,7 @@ import logging import os import re +import time from dataclasses import dataclass, field from typing import Any @@ -282,6 +283,10 @@ class AnomalyRunResult: response_id: str | None = None tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from + # turn_wall_clock_sec above, which is only the final triggering turn. + run_latency_s: float = 0.0 @dataclass @@ -360,6 +365,7 @@ def run_agentic_anomaly_detection( ) def _run_once(conv_id: str) -> AnomalyRunResult: + run_started = time.monotonic() viz_args: dict | None = None execute_result: dict | None = None turn_wall_clock_sec: float | None = None @@ -432,6 +438,7 @@ def _accumulate(result: ChatResult) -> None: response_id=response_id, tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, + run_latency_s=time.monotonic() - run_started, ) try: @@ -630,6 +637,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = len(summary.run_results) + exc.best_run_latency_s = best.run_latency_s raise exc return AgenticEvalOutcome( @@ -639,4 +647,5 @@ def _write_scores(ctx: RunTraceContext) -> None: conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py index 695295a7b..349adcf77 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py @@ -19,6 +19,7 @@ import json import os import re +import time from dataclasses import dataclass, field from datetime import date from typing import ClassVar, Literal @@ -593,6 +594,11 @@ class ConversationResult: stalled_turns: int = 0 mode: ConversationMode = "legacy" judge_model: str | None = None + # This conversation's own wall time, across every turn -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). There is no + # K-of-N selection for this kind (see run_agentic_conversation's docstring), so + # this is simply the one conversation's own time, not a "best of" pick. + run_latency_s: float = 0.0 def _metric_creations(tool_call_events: list[ToolCallEvent]) -> list[tuple[str, bool]]: @@ -707,6 +713,7 @@ def run_agentic_conversation( agent sees no history. It is the no-memory baseline a context score is calibrated against: context-dependent turns are expected to fail there. """ + run_started = time.monotonic() resolved_mode = resolve_conversation_mode(mode) if fresh_conversation_per_turn and initial_conversation_id is not None: raise ValueError("fresh_conversation_per_turn needs conversations this run creates itself") @@ -1011,6 +1018,7 @@ def run_agentic_conversation( stalled_turns=stalled, mode=resolved_mode, judge_model=judge.model_name if judged else None, + run_latency_s=time.monotonic() - run_started, ) @@ -1216,6 +1224,7 @@ def _write_scores(ctx: RunTraceContext) -> None: # for. Saying so explicitly stops the report claiming K runs that never happened. exc.runs_passed = 0 exc.runs_effective = 1 + exc.best_run_latency_s = result.run_latency_s raise exc return AgenticEvalOutcome( reasoning_steps=result.reasoning_steps, @@ -1224,4 +1233,5 @@ def _write_scores(ctx: RunTraceContext) -> None: detail=detail, runs_passed=1, runs_effective=1, + best_run_latency_s=result.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py index 9e017d57a..f33cc9dae 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py @@ -38,7 +38,7 @@ build_latency_breakdown, shift_and_index_events, ) -from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings +from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings _DEFAULT_K = 1 # Matches kda_skill and visualization. The reply this loop sends is built from the expectation @@ -1137,6 +1137,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.conversation_id = best.conversation_id exc.response_id = best.response_id exc.timings = item_timings + exc.best_run_latency_s = run_latency_s(best.timings) exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective @@ -1149,4 +1150,5 @@ def _write_scores(ctx: RunTraceContext) -> None: response_id=best.response_id, detail=detail, timings=item_timings, + best_run_latency_s=run_latency_s(best.timings), ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py index 8c8fb3727..6db82dbda 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py @@ -34,7 +34,7 @@ ToolCallEvent, build_latency_breakdown, ) -from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings +from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings _DEFAULT_K = 1 @@ -318,6 +318,7 @@ def _write_scores(ctx: RunTraceContext) -> None: + " | ".join(unscored) ) exc.timings = item_timings + exc.best_run_latency_s = run_latency_s(summary.best.timings) raise exc runs_passed = sum(1 for r in summary.scored_run_results if r.passed) @@ -344,6 +345,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.conversation_id = best.conversation_id exc.response_id = best.response_id exc.timings = item_timings + exc.best_run_latency_s = run_latency_s(best.timings) exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective @@ -356,4 +358,5 @@ def _write_scores(ctx: RunTraceContext) -> None: response_id=best.response_id, detail=detail, timings=item_timings, + best_run_latency_s=run_latency_s(best.timings), ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py index f3cb43635..bed578c35 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py @@ -3,6 +3,7 @@ from __future__ import annotations +import time from dataclasses import dataclass, field from gooddata_eval.core.agentic._gate import ( @@ -83,6 +84,9 @@ class GuardrailResult: # Set when the judge returned something unreadable for THIS run. Excluded from pass@K # and from Langfuse scoring rather than counted as a failure -- see score_run. judge_error: str | None = None + # This run's own wall time (agent call + its grading) -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -117,6 +121,7 @@ def _run_single_guardrail( Extracted so the two call sites below (the first conversation, which may be supplied, and the remaining K-1) cannot drift -- they had already duplicated the whole body once. """ + run_started = time.monotonic() chat_result = client.send_message(conversation_id, question) actual_output = render_answer_text(chat_result) verdict = score_run(judge, input=question, expected_output=expected_output, actual_output=actual_output) @@ -131,6 +136,7 @@ def _run_single_guardrail( tool_call_events=list(chat_result.tool_call_events or []), reasoning_step_events=list(chat_result.reasoning_step_events or []), judge_error=verdict.error, + run_latency_s=time.monotonic() - run_started, ) @@ -314,6 +320,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective + exc.best_run_latency_s = best.run_latency_s raise exc return AgenticEvalOutcome( runs_passed=runs_passed, @@ -322,4 +329,5 @@ def _write_scores(ctx: RunTraceContext) -> None: conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py index 9e0ced8a6..7b29bf624 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py @@ -5,6 +5,7 @@ import logging import os +import time from dataclasses import dataclass, field import httpx @@ -198,6 +199,10 @@ class KdaRunResult: # separate a refusal from a run that hit max_iterations while still on track. exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED turns_used: int = 0 + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from + # turn_wall_clock_sec above, which is only the final create-triggering turn. + run_latency_s: float = 0.0 @dataclass @@ -260,6 +265,7 @@ def run_agentic_kda_skill( ) def _run_once(conv_id: str) -> KdaRunResult: + run_started = time.monotonic() create_args: dict | None = None execute_result: dict | None = None turn_wall_clock_sec: float | None = None @@ -367,6 +373,7 @@ def _accumulate(result: ChatResult) -> None: reasoning_step_events=all_reasoning_step_events, exit_reason=exit_reason, turns_used=turns_used, + run_latency_s=time.monotonic() - run_started, ) try: @@ -552,6 +559,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective + exc.best_run_latency_s = best.run_latency_s raise exc return AgenticEvalOutcome( runs_passed=runs_passed, @@ -560,4 +568,5 @@ def _write_scores(ctx: RunTraceContext) -> None: conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py index d00abe942..e2f95109e 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py @@ -41,7 +41,7 @@ shift_and_index_events, timeline_detail, ) -from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings +from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings try: from openai import OpenAI as _OpenAI @@ -575,6 +575,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.conversation_id = best.conversation_id exc.response_id = best.response_id exc.timings = item_timings + exc.best_run_latency_s = run_latency_s(best.timings) exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective @@ -587,4 +588,5 @@ def _write_scores(ctx: RunTraceContext) -> None: response_id=best.response_id, detail=detail, timings=item_timings, + best_run_latency_s=run_latency_s(best.timings), ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py index ac70deeb7..95fd9ef8a 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py @@ -3,6 +3,7 @@ from __future__ import annotations +import time from dataclasses import dataclass, field from gooddata_eval.core.agentic._gate import ( @@ -75,6 +76,9 @@ class SearchResult: response_id: str | None = None tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + # This run's own wall time -- mirrors the single-shot path's best_run_latency_s + # (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -107,6 +111,7 @@ def run_agentic_search_tool( try: conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() try: + run_started = time.monotonic() chat_result = client.send_message(conv_id_0, question) tcs = chat_result.tool_call_events or [] selected = _tool_selection(tcs) @@ -121,6 +126,7 @@ def run_agentic_search_tool( response_id=chat_result.response_id, tool_call_events=list(chat_result.tool_call_events or []), reasoning_step_events=list(chat_result.reasoning_step_events or []), + run_latency_s=time.monotonic() - run_started, ) ) finally: @@ -130,6 +136,7 @@ def run_agentic_search_tool( for _ in range(1, k): conv_id = client.create_conversation() try: + run_started = time.monotonic() chat_result = client.send_message(conv_id, question) tcs = chat_result.tool_call_events or [] selected = _tool_selection(tcs) @@ -144,6 +151,7 @@ def run_agentic_search_tool( response_id=chat_result.response_id, tool_call_events=list(chat_result.tool_call_events or []), reasoning_step_events=list(chat_result.reasoning_step_events or []), + run_latency_s=time.monotonic() - run_started, ) ) finally: @@ -275,6 +283,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective + exc.best_run_latency_s = best.run_latency_s raise exc return AgenticEvalOutcome( runs_passed=runs_passed, @@ -283,4 +292,5 @@ def _write_scores(ctx: RunTraceContext) -> None: conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py index 51b60f51c..9ad64b722 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py @@ -8,6 +8,7 @@ from __future__ import annotations import os +import time from dataclasses import dataclass, field from gooddata_eval.core.agentic._gate import ( @@ -70,6 +71,9 @@ class RunResult: # Why the simulated-user loop stopped. visualization_created=False alone cannot separate # a refusal from a run that hit max_iterations while still on track -- see LoopExit. exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -191,6 +195,7 @@ def _execute_single_run( max_iterations: int = _DEFAULT_MAX_ITERATIONS, ) -> RunResult: """Drive one full multi-turn conversation and evaluate the result.""" + run_started = time.monotonic() total_turns = 0 total_steps = 0 all_tool_call_events: list[ToolCallEvent] = [] @@ -287,6 +292,7 @@ def _execute_single_run( tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, exit_reason=exit_reason, + run_latency_s=time.monotonic() - run_started, ) @@ -532,6 +538,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = runs_effective + exc.best_run_latency_s = best.run_latency_s raise exc return AgenticEvalOutcome( runs_passed=runs_passed, @@ -540,4 +547,5 @@ def _write_scores(ctx: RunTraceContext) -> None: conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py index 9a01b1575..d51ac6fde 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py @@ -22,6 +22,7 @@ import logging import os +import time from dataclasses import dataclass, field from typing import Any @@ -201,6 +202,10 @@ class WhatIfRunResult: response_id: str | None = None tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from + # turn_wall_clock_sec above, which is only the scenario-execution turn. + run_latency_s: float = 0.0 @dataclass @@ -309,6 +314,7 @@ def run_agentic_what_if( ) def _run_once(conv_id: str) -> WhatIfRunResult: + run_started = time.monotonic() create_args: dict | None = None execute_result: dict | None = None turn_wall_clock_sec: float | None = None @@ -381,6 +387,7 @@ def _accumulate(result: ChatResult) -> None: response_id=response_id, tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, + run_latency_s=time.monotonic() - run_started, ) try: @@ -581,6 +588,7 @@ def _write_scores(ctx: RunTraceContext) -> None: exc.detail = detail exc.runs_passed = runs_passed exc.runs_effective = len(summary.run_results) + exc.best_run_latency_s = best.run_latency_s raise exc return AgenticEvalOutcome( @@ -590,4 +598,5 @@ def _write_scores(ctx: RunTraceContext) -> None: conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py b/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py index d7e9ddc5f..4c9ee2dca 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py @@ -49,6 +49,7 @@ class JudgeResponseError(RuntimeError): # rather than attached loosely, because the runner reads it off the exception to report # what an unevaluable item still cost. timings: PhaseTimings + best_run_latency_s: float | None def _message_content(response: Any) -> str | None: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/models.py b/packages/gooddata-eval/src/gooddata_eval/core/models.py index 142fc20f6..79067e42e 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/models.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/models.py @@ -408,6 +408,7 @@ class AgenticAssertionError(AssertionError): timings: PhaseTimings runs_passed: int runs_effective: int + best_run_latency_s: float | None class AgenticEvalOutcome(BaseModel): @@ -433,6 +434,12 @@ class AgenticEvalOutcome(BaseModel): # ran -- agentic_conversation drives its fixture once whatever --runs says. runs_passed: int = 0 runs_effective: int = 0 + # The rank-selected run's own wall time -- mirrors the single-shot path's + # ItemReport.best_run_latency_s (see core/runner.py's _run_one_item), which the + # agentic path never populated because its K runs happen inside the evaluator, + # not in a loop the runner itself can time. None for a kind that has not yet been + # wired to measure it (see AgenticAssertionError.best_run_latency_s, same default). + best_run_latency_s: float | None = None class SummaryInput(BaseModel): diff --git a/packages/gooddata-eval/src/gooddata_eval/core/timing.py b/packages/gooddata-eval/src/gooddata_eval/core/timing.py index a3c05aef1..f9d6e2812 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/timing.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/timing.py @@ -72,3 +72,13 @@ def as_dict(self) -> dict[str, float]: def sum_timings(timings: list[PhaseTimings]) -> PhaseTimings: """Total across a list of per-run timings (empty list -> all zeroes).""" return sum(timings, PhaseTimings()) + + +def run_latency_s(timings: PhaseTimings) -> float: + """One run's own wall time, for a single run's ``PhaseTimings``. + + Excludes ``langfuse_s`` (deliberately off the critical path -- see + ``PhaseTimings.langfuse_s``) so this stays comparable to the single-shot path's + ``best_run_latency_s``, which times only the agent call and its grading. + """ + return timings.agent_s + timings.judge_s + timings.simulated_user_s diff --git a/packages/gooddata-eval/tests/test_agentic_general_question.py b/packages/gooddata-eval/tests/test_agentic_general_question.py index 06609f21d..cd9db187e 100644 --- a/packages/gooddata-eval/tests/test_agentic_general_question.py +++ b/packages/gooddata-eval/tests/test_agentic_general_question.py @@ -400,6 +400,9 @@ def test_evaluate_general_question_aggregates_timings_across_k_runs(): assert outcome.timings.agent_s == 5.0 # 2.0 + 3.0 assert outcome.timings.judge_s == 1.5 # 1.0 + 0.5 + # best_run_latency_s is the rank-selected run's own time (agent_s=2.0 + judge_s=1.0 + # from the first, tied-best run), not the item total above (both runs summed). + assert outcome.best_run_latency_s == 3.0 def test_evaluate_general_question_attaches_timings_to_the_failure_exception(): @@ -423,6 +426,7 @@ def test_evaluate_general_question_attaches_timings_to_the_failure_exception(): assert exc_info.value.timings.agent_s == 4.0 assert exc_info.value.timings.judge_s == 2.0 + assert exc_info.value.best_run_latency_s == 6.0 def test_trace_link_window_is_pinned_at_submit_time_not_when_the_task_runs(): diff --git a/packages/gooddata-eval/tests/test_agentic_metric_skill.py b/packages/gooddata-eval/tests/test_agentic_metric_skill.py index 1a747c437..93ef45b4f 100644 --- a/packages/gooddata-eval/tests/test_agentic_metric_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_metric_skill.py @@ -754,6 +754,7 @@ def test_evaluate_metric_skill_surfaces_timings_on_the_outcome(): assert outcome.timings.agent_s == 3.0 assert outcome.timings.simulated_user_s == 0.0 + assert outcome.best_run_latency_s == 3.0 def test_no_timer_output_by_default(monkeypatch, capsys): diff --git a/packages/gooddata-eval/tests/test_agentic_runner.py b/packages/gooddata-eval/tests/test_agentic_runner.py index 2221586d5..1d38b345e 100644 --- a/packages/gooddata-eval/tests/test_agentic_runner.py +++ b/packages/gooddata-eval/tests/test_agentic_runner.py @@ -341,6 +341,54 @@ def test_run_agentic_items_records_phase_timings_when_the_item_fails(): assert report.items[0].judge_latency_s == 2.0 +def test_run_agentic_items_records_best_run_latency_on_success(): + # best_run_latency_s was never populated on the agentic path -- the K-run loop and + # best-of-K selection happen inside the evaluator, not in a loop the runner itself + # times (unlike _run_one_item on the single-shot path). + outcome = AgenticEvalOutcome( + conversation_id="c1", + detail={"alert_created": True}, + best_run_latency_s=4.25, + ) + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", return_value=outcome): + report = run_agentic_items( + [_timed_item()], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + + assert report.items[0].best_run_latency_s == 4.25 + + +def test_run_agentic_items_records_best_run_latency_when_the_item_fails(): + exc = AlertSkillAssertionError("nope") + exc.detail = {"alert_created": False} + exc.best_run_latency_s = 9.5 + + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", side_effect=exc): + report = run_agentic_items( + [_timed_item()], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + + assert report.items[0].pass_at_k is False + assert report.items[0].best_run_latency_s == 9.5 + + +def test_an_errored_item_without_best_run_latency_keeps_the_none_default(): + # Kinds not yet wired to measure it (or any exception with no such attribute) must + # not report an invented number -- None stays None, not 0.0. + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", side_effect=RuntimeError("boom")): + report = run_agentic_items([_timed_item()], host="h", token="t", workspace_id="ws1", run_ts="2026-01-01") + + assert report.items[0].best_run_latency_s is None + + def test_a_pending_trace_link_does_not_block_the_next_item(): """The overlap claim, proved by construction rather than by a stopwatch. diff --git a/packages/gooddata-eval/tests/test_timing.py b/packages/gooddata-eval/tests/test_timing.py index 3b31a8411..de0951af8 100644 --- a/packages/gooddata-eval/tests/test_timing.py +++ b/packages/gooddata-eval/tests/test_timing.py @@ -9,6 +9,7 @@ TIMERS_ENV_VAR, PhaseTimings, log_timer, + run_latency_s, sum_timings, timers_enabled, ) @@ -91,3 +92,10 @@ def test_as_dict_rounds_every_phase(): def test_sum_timings_of_nothing_is_all_zeroes(): assert sum_timings([]) == PhaseTimings() + + +def test_run_latency_s_excludes_langfuse(): + # langfuse_s is deliberately off the critical path (see PhaseTimings docstring), so a + # trace-link slowdown must not inflate the run latency a worst-case table sorts on. + timings = PhaseTimings(agent_s=2.0, judge_s=1.0, simulated_user_s=0.5, langfuse_s=100.0) + assert run_latency_s(timings) == 3.5 From 4216bede1b276f927914434d94a18b4a23ba374e Mon Sep 17 00:00:00 2001 From: Peter Tomko Date: Mon, 5 Oct 2026 22:18:31 +0200 Subject: [PATCH 2/4] fix(gooddata-eval): propagate best_run_latency_s on guardrail's no-verdict path CodeRabbit catch on #1845: the JudgeResponseError raised when no run's judge verdict was readable at all left best_run_latency_s unset, unlike the equivalent branch in general_question.py. Co-Authored-By: Claude Sonnet 5 --- .../gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py index bed578c35..bc5a4f4d0 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py @@ -290,10 +290,12 @@ def _write_scores(ctx: RunTraceContext) -> None: if not summary.scored_run_results: # No readable verdict for any run: an error, not K failures. Raised after the # trace link is queued so whatever the agent did is still linked. - raise JudgeResponseError( + exc = JudgeResponseError( f"judge returned no readable verdict for any of the {len(summary.run_results)} run(s): " + " | ".join(unscored) ) + exc.best_run_latency_s = summary.best.run_latency_s + raise exc runs_passed = sum(1 for r in summary.scored_run_results if r.passed) runs_effective = len(summary.run_results) From c41f542f207d22b90e00f290f4b23281eb06bde2 Mon Sep 17 00:00:00 2001 From: Peter Tomko Date: Mon, 5 Oct 2026 22:46:40 +0200 Subject: [PATCH 3/4] refactor(gooddata-eval): centralize agentic outcome fields; fix exit_reason and tool_calls gaps (#1846) Comprehensive analysis of the 11 agentic evaluators (alert_skill, anomaly_detection, conversation, dashboard_skill, general_question, guardrail, kda_skill, metric_skill, search_tool, visualization, what_if) found the same unfinished-rollout pattern that caused best_run_latency_s (#1845) recurring twice more, live in production data: 1. exit_reason (LoopExit) -- invented specifically so a run that hit max_iterations stops looking identical to a genuine refusal -- was wired into 5 of 8 multi-turn kinds. Missing from anomaly_detection, dashboard_skill, what_if. All three now track it through every break point in their run loop (SUCCESS/AGENT_SILENT/ CHAT_ERROR/SIMULATED_USER_FAILED/BUDGET_EXHAUSTED default), mirroring the exact pattern already proven in kda_skill.py/alert_skill.py. 2. detail's tool_calls key -- timeline_detail() builds both latency_breakdown and tool_calls from the same events so they stay index-aligned, but roughly half the evaluators called build_latency_breakdown() directly and silently never got a tool_calls key. Every evaluator's detail dict now goes through one shared `agentic_detail()` call, so this can't happen again by omission. 3. The root cause of 1 and 2 (and best_run_latency_s before them): every evaluator hand-wrote the same ~7-10 line block copying reasoning_steps/conversation_id/ response_id/detail/runs_passed/runs_effective/(timings)/best_run_latency_s either onto a raised exception or into the returned AgenticEvalOutcome -- 11 independent copies of the same logic, so a new universal field needed 11 edits, and nothing failed loudly when one was missed (AgenticAssertionError's own docstring already documents this happening once before, to `.timings`). New core/agentic/_outcome.py centralizes it: `agentic_detail()` for the detail dict, `raise_agentic_failure()`/`agentic_success()` for the common fields. All 11 evaluators now call these instead of hand-rolling. Per-evaluator duplicate field assignments: ~85 lines total before, 3 after (the 2 remaining are general_question's and guardrail's "no judge verdict at all" branches, which raise a bare JudgeResponseError -- not an AgenticAssertionError subclass, so they can't route through raise_agentic_failure; left as direct attribute sets, same as before). Explicitly NOT addressed here (see PR description for why): - agentic_dashboard_summary not existing on master at all -- it's part of a much larger (2000+ line) unmerged integration branch with its own in-flight PRs; porting it here would duplicate/conflict with that separate effort. - A ~9% gap in agentic_guardrail's .reasoning.json sidecar capture for runs that did have reasoning steps -- traced as far as ruling out config differences and SSE-level field divergence (reasoningSteps/reasoningStepEvents are built from the same accumulator list, so they can't diverge there), but the remaining cause needs a live repro with real reasoning content, not static source reading. Verified: full gooddata-eval suite (1382 passed), ruff lint/format, and ty type-check all clean. New/extended tests assert real exit_reason values (not just that nothing broke) for every break point added, plus tool_calls presence, across all three newly-covered kinds. Co-authored-by: Claude Sonnet 5 --- .../gooddata_eval/core/agentic/_outcome.py | 109 ++++++++++++++++++ .../gooddata_eval/core/agentic/alert_skill.py | 61 +++++----- .../core/agentic/anomaly_detection.py | 78 ++++++++----- .../core/agentic/conversation.py | 55 ++++----- .../core/agentic/dashboard_skill.py | 60 ++++++---- .../core/agentic/general_question.py | 45 ++++---- .../gooddata_eval/core/agentic/guardrail.py | 41 +++---- .../gooddata_eval/core/agentic/kda_skill.py | 55 ++++----- .../core/agentic/metric_skill.py | 53 ++++----- .../gooddata_eval/core/agentic/search_tool.py | 41 +++---- .../core/agentic/visualization.py | 41 +++---- .../src/gooddata_eval/core/agentic/what_if.py | 80 ++++++++----- .../tests/test_agentic_anomaly_detection.py | 12 +- .../tests/test_agentic_dashboard_skill.py | 7 +- .../tests/test_agentic_general_question.py | 2 + .../tests/test_agentic_kda_skill.py | 2 + .../tests/test_agentic_search_tool.py | 2 + .../tests/test_agentic_what_if.py | 14 ++- 18 files changed, 483 insertions(+), 275 deletions(-) create mode 100644 packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py new file mode 100644 index 000000000..9c35413e8 --- /dev/null +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py @@ -0,0 +1,109 @@ +# (C) 2026 GoodData Corporation +"""The common tail of every ``evaluate_agentic_*`` function. + +Before this module, each of the agentic evaluators hand-wrote the same block: copy +``reasoning_steps``/``conversation_id``/``response_id``/``detail``/``runs_passed``/ +``runs_effective``/``best_run_latency_s`` (and sometimes ``timings``) either onto a +raised exception or into the returned ``AgenticEvalOutcome``. Eleven independent copies +of the same ~7 lines meant a new universal field needed eleven edits, not one, and +nothing failed loudly when a copy was missed -- exactly what happened to ``timings`` +(see ``AgenticAssertionError``'s docstring: "while it lived in eight copies, two +declared timings and six did not") and again to ``best_run_latency_s`` (every agentic +kind silently reported it as ``null`` until this was noticed and fixed kind-by-kind). + +``agentic_detail`` has the same motivation for the ``detail`` dict's timeline fields: +``timeline_detail`` builds both ``latency_breakdown`` and ``tool_calls`` from the same +events so they stay index-aligned, but roughly half the evaluators called +``build_latency_breakdown`` directly and silently never got a ``tool_calls`` key. +""" + +from __future__ import annotations + +from typing import Any, NoReturn + +from gooddata_eval.core.models import ( + AgenticAssertionError, + AgenticEvalOutcome, + ReasoningStepEvent, + ToolCallEvent, + timeline_detail, +) +from gooddata_eval.core.timing import PhaseTimings + +__all__ = ["agentic_detail", "agentic_success", "raise_agentic_failure"] + + +def agentic_detail( + tool_call_events: list[ToolCallEvent], + reasoning_step_events: list[ReasoningStepEvent] | None, + **kind_specific: Any, +) -> dict: + """A kind's full ``detail`` dict: its own fields plus the universal timeline ones. + + ``kind_specific`` comes first in the merge so a kind can never accidentally shadow + ``latency_breakdown``/``tool_calls`` with a same-named field of its own. + """ + return {**kind_specific, **timeline_detail(tool_call_events, reasoning_step_events)} + + +def raise_agentic_failure( + exception_cls: type[AgenticAssertionError], + message: str, + *, + reasoning_steps: list[str], + conversation_id: str, + response_id: str | None, + detail: dict, + runs_passed: int, + runs_effective: int, + best_run_latency_s: float | None, + timings: PhaseTimings | None = None, +) -> NoReturn: + """Build ``exception_cls(message)`` with every common field attached, and raise it. + + ``exception_cls`` must be an ``AgenticAssertionError`` subclass -- that base class is + what declares these fields as legal targets (see its docstring). A kind whose own + "no verdict at all" branch raises a bare ``JudgeResponseError`` instead (not a subclass) + does not go through this helper for that branch. + + ``timings`` stays optional: only the three kinds that track per-run ``PhaseTimings`` + (general_question, metric_skill, dashboard_skill) pass one -- the rest keep their + own ``PhaseTimings()`` zero default, same as before this helper existed. + """ + exc = exception_cls(message) + exc.reasoning_steps = reasoning_steps + exc.conversation_id = conversation_id + exc.response_id = response_id + exc.detail = detail + exc.runs_passed = runs_passed + exc.runs_effective = runs_effective + exc.best_run_latency_s = best_run_latency_s + if timings is not None: + exc.timings = timings + raise exc + + +def agentic_success( + *, + reasoning_steps: list[str], + conversation_id: str | None, + response_id: str | None, + detail: dict, + runs_passed: int, + runs_effective: int, + best_run_latency_s: float | None, + timings: PhaseTimings | None = None, +) -> AgenticEvalOutcome: + """The success-path mirror of ``raise_agentic_failure`` -- same fields, same shape.""" + kwargs: dict[str, Any] = { + "reasoning_steps": reasoning_steps, + "conversation_id": conversation_id, + "response_id": response_id, + "detail": detail, + "runs_passed": runs_passed, + "runs_effective": runs_effective, + "best_run_latency_s": best_run_latency_s, + } + if timings is not None: + kwargs["timings"] = timings + return AgenticEvalOutcome(**kwargs) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py index bb59fdf45..45c9d1bc1 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py @@ -21,6 +21,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -40,7 +41,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) try: @@ -951,28 +951,30 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.eval - detail = { - "alert_created": ev.alert_created, - "operator_correct": ev.operator_correct, - "threshold_correct": ev.threshold_correct, - "trigger_correct": ev.trigger_correct, - "filters_correct": ev.filters_correct, - "metric_correct": ev.metric_correct, - "recipients_correct": ev.recipients_correct, - "attributes_correct": ev.attributes_correct, - "granularity_correct": ev.granularity_correct, - "actual_alert_arguments": best.actual_alert_arguments, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + alert_created=ev.alert_created, + operator_correct=ev.operator_correct, + threshold_correct=ev.threshold_correct, + trigger_correct=ev.trigger_correct, + filters_correct=ev.filters_correct, + metric_correct=ev.metric_correct, + recipients_correct=ev.recipients_correct, + attributes_correct=ev.attributes_correct, + granularity_correct=ev.granularity_correct, + actual_alert_arguments=best.actual_alert_arguments, # Why the loop stopped. alert_created=False alone cannot tell a refusal from a run # that hit max_iterations while still on track -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) - exc = AlertSkillAssertionError( + raise_agentic_failure( + AlertSkillAssertionError, f"Alert skill assertion failed. {gate_note} strict_pass={ev.strict_pass}. " f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, " f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, " @@ -980,22 +982,21 @@ def _write_scores(ctx: RunTraceContext) -> None: f"recipients_correct={ev.recipients_correct}, " f"attributes_correct={ev.attributes_correct}, " f"granularity_correct={ev.granularity_correct}. " - f"Actual args: {best.actual_alert_arguments}" + f"Actual args: {best.actual_alert_arguments}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py index daa603064..a2b4050ee 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py @@ -40,6 +40,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -56,9 +57,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -287,6 +288,10 @@ class AnomalyRunResult: # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from # turn_wall_clock_sec above, which is only the final triggering turn. run_latency_s: float = 0.0 + # Why the simulated-user loop stopped -- see LoopExit. `triggered=False` alone cannot + # separate a refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -391,6 +396,10 @@ def _accumulate(result: ChatResult) -> None: all_tool_call_events.extend(result.tool_call_events or []) all_reasoning_step_events.extend(result.reasoning_step_events or []) + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED + for iteration in range(max_iterations): try: chat_result = client.send_message(conv_id, current_question) @@ -403,6 +412,7 @@ def _accumulate(result: ChatResult) -> None: _accumulate(partial) viz_args, execute_result = _extract_anomaly_calls(all_tool_call_events) turn_completed = False + exit_reason = LoopExit.CHAT_ERROR break reasoning_steps.extend(chat_result.reasoning_steps or []) response_id = chat_result.response_id or response_id @@ -416,8 +426,10 @@ def _accumulate(result: ChatResult) -> None: if execute_result is not None: # The turn that ran the detection, not an earlier disambiguation turn. turn_wall_clock_sec = chat_result.turn_wall_clock_sec + exit_reason = LoopExit.SUCCESS break if not response_text: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -426,6 +438,7 @@ def _accumulate(result: ChatResult) -> None: disambiguated = True except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run _log.warning("Simulated anomaly user reply failed for conversation %s: %s", conv_id, exc) + exit_reason = LoopExit.SIMULATED_USER_FAILED break return AnomalyRunResult( @@ -434,6 +447,7 @@ def _accumulate(result: ChatResult) -> None: actual_visualization=viz_args, actual_execute_result=execute_result, turn_wall_clock_sec=turn_wall_clock_sec, + exit_reason=exit_reason, reasoning_steps=reasoning_steps, response_id=response_id, tool_call_events=all_tool_call_events, @@ -487,25 +501,29 @@ class AnomalyDetectionAssertionError(AgenticAssertionError): def _detail(best: AnomalyRunResult) -> dict[str, Any]: ev = best.evaluation - return { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "metric_correct": ev.metric_correct, - "granularity_correct": ev.granularity_correct, + return agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + metric_correct=ev.metric_correct, + granularity_correct=ev.granularity_correct, # Which content checks the fixture pinned -- without it a run that verified nothing # reads the same as one where everything matched. - "asserted": ev.asserted, - "disambiguated": ev.disambiguated, - "actual_metrics": sorted(_metric_uris(best.actual_visualization)), - "actual_granularity": _inferred_granularity(best.actual_visualization), + asserted=ev.asserted, + disambiguated=ev.disambiguated, + actual_metrics=sorted(_metric_uris(best.actual_visualization)), + actual_granularity=_inferred_granularity(best.actual_visualization), # Reported, never asserted: whether a real series contains anomalies is a property # of the data, so a fixture demanding some would fail on the next warehouse refresh. - "anomaly_point_count": _point_count(best.actual_execute_result), - "actual_execute_result": best.actual_execute_result, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + anomaly_point_count=_point_count(best.actual_execute_result), + actual_execute_result=best.actual_execute_result, + # Why the loop stopped. triggered=False alone cannot tell a refusal from a run + # that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) def evaluate_agentic_anomaly_detection( @@ -630,22 +648,24 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Analysed {detail['actual_metrics']} at {detail['actual_granularity']}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = AnomalyDetectionAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = len(summary.run_results) - exc.best_run_latency_s = best.run_latency_s - raise exc - - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=len(summary.run_results), + raise_agentic_failure( + AnomalyDetectionAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, + ) + + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py index 349adcf77..e41cc271f 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py @@ -41,6 +41,7 @@ summarize_visualizations, ) from gooddata_eval.core.agentic._gate import log_gate_scores +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -66,7 +67,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.scoring import ( check_filters, @@ -1023,21 +1023,22 @@ def run_agentic_conversation( def _conversation_detail(result: ConversationResult) -> dict: - return { - "mode": result.mode, - "full_skill_coverage": result.full_skill_coverage, - "conversation_success": result.conversation_success, - "context_success": result.context_success, - "context_kept_rate": result.context_kept_rate, - "turns_before_first_break": result.turns_before_first_break, - "lost_context_clarifications": result.lost_context_clarifications, - "stalled_turns": result.stalled_turns, - "total_clarification_turns": result.total_clarification_turns, - "max_clarification_turns": result.max_clarification_turns, - "judge_model": result.judge_model, - "turns": [tr.detail() for tr in result.turn_results], - **timeline_detail(result.tool_call_events, result.reasoning_step_events), - } + return agentic_detail( + result.tool_call_events, + result.reasoning_step_events, + mode=result.mode, + full_skill_coverage=result.full_skill_coverage, + conversation_success=result.conversation_success, + context_success=result.context_success, + context_kept_rate=result.context_kept_rate, + turns_before_first_break=result.turns_before_first_break, + lost_context_clarifications=result.lost_context_clarifications, + stalled_turns=result.stalled_turns, + total_clarification_turns=result.total_clarification_turns, + max_clarification_turns=result.max_clarification_turns, + judge_model=result.judge_model, + turns=[tr.detail() for tr in result.turn_results], + ) class ConversationAssertionError(AgenticAssertionError): @@ -1215,18 +1216,20 @@ def _write_scores(ctx: RunTraceContext) -> None: f"full_skill_coverage={result.full_skill_coverage}. " f"Failed turns: {legacy_failed}" ) - exc = ConversationAssertionError(message) - exc.reasoning_steps = result.reasoning_steps - exc.conversation_id = result.conversation_id - exc.response_id = result.response_id - exc.detail = detail # This kind takes no k and drives its fixture exactly once, whatever --runs asks # for. Saying so explicitly stops the report claiming K runs that never happened. - exc.runs_passed = 0 - exc.runs_effective = 1 - exc.best_run_latency_s = result.run_latency_s - raise exc - return AgenticEvalOutcome( + raise_agentic_failure( + ConversationAssertionError, + message, + reasoning_steps=result.reasoning_steps, + conversation_id=result.conversation_id, + response_id=result.response_id, + detail=detail, + runs_passed=0, + runs_effective=1, + best_run_latency_s=result.run_latency_s, + ) + return agentic_success( reasoning_steps=result.reasoning_steps, conversation_id=result.conversation_id, response_id=result.response_id, diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py index f33cc9dae..29cc1892d 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py @@ -17,6 +17,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -33,9 +34,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings @@ -664,6 +665,11 @@ class DashboardRunResult: tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) timings: PhaseTimings = field(default_factory=PhaseTimings) + # Why the loop stopped -- see LoopExit. A dashboard not produced alone cannot tell a + # refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). No CHAT_ERROR/SIMULATED_USER_FAILED member here: this + # loop has neither a send_message try/except nor a simulated-user call to fail. + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -864,6 +870,11 @@ def _execute_single_dashboard_run( turn_offset = 0.0 tool_index_offset = 0 reasoning_index_offset = 0 + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() (or the is_edit "no reply of its own" branch below, which + # is the same shape of outcome -- the agent answered, just not with the dashboard tool) + # is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED for iteration in range(max_iterations): turns += 1 @@ -903,10 +914,12 @@ def _execute_single_dashboard_run( dashboard_part = _extract_dashboard_part(chat_result, "dashboard") if is_edit: patch_part = _extract_dashboard_part(chat_result, _PATCH_TYPE) + exit_reason = LoopExit.SUCCESS break response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) if not response_text and not chat_result.tool_call_events: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -941,6 +954,7 @@ def _execute_single_dashboard_run( tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, timings=timings, + exit_reason=exit_reason, ) @@ -1100,13 +1114,17 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail: dict[str, Any] = { + detail: dict[str, Any] = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, **best.evaluation.strict_checks, **best.evaluation.diagnostics, - "failures": best.evaluation.failures, - "notes": best.evaluation.notes, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + failures=best.evaluation.failures, + notes=best.evaluation.notes, + # Why the loop stopped. A dashboard not produced alone cannot tell a refusal from + # a run that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -1127,28 +1145,28 @@ def _write_scores(ctx: RunTraceContext) -> None: " one feature flag registers both, so check that before the model." ) notes = "; ".join(best.evaluation.notes) - exc = DashboardSkillAssertionError( + raise_agentic_failure( + DashboardSkillAssertionError, f"Dashboard skill assertion failed. {gate_note}{skill_note} " f"Checks: {best.evaluation.strict_checks}. " f"Failures: {'; '.join(best.evaluation.failures) or 'none reported'}." - + (f" Notes (not scored): {notes}." if notes else "") + + (f" Notes (not scored): {notes}." if notes else ""), + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.best_run_latency_s = run_latency_s(best.timings) - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, - timings=item_timings, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py index 6db82dbda..b54e772e5 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py @@ -14,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -32,7 +33,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, ) from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings @@ -325,38 +325,39 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail = { - "judge_passed": best.passed, - "judge_reasoning": best.reasoning, - "actual_output": best.actual_output, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + judge_passed=best.passed, + judge_reasoning=best.reasoning, + actual_output=best.actual_output, # Only present when it happened, so the usual JSON shape is unchanged. A # pass@K computed over fewer runs than --runs asked for is a weaker result and # the report has to say so. **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), - } + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - exc = GeneralQuestionAssertionError( - f"General question assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}" + raise_agentic_failure( + GeneralQuestionAssertionError, + f"General question assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.best_run_latency_s = run_latency_s(best.timings) - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, - timings=item_timings, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py index bc5a4f4d0..4c1d9d2fd 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py @@ -14,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -33,7 +34,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - timeline_detail, ) _DEFAULT_K = 1 @@ -301,35 +301,36 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail = { - "judge_passed": best.passed, - "judge_reasoning": best.reasoning, - "actual_output": best.actual_output, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + judge_passed=best.passed, + judge_reasoning=best.reasoning, + actual_output=best.actual_output, # Only present when it happened, so the usual JSON shape is unchanged. A # pass@K over fewer runs than --runs asked for is a weaker result. **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), - } + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - exc = GuardrailAssertionError( - f"Guardrail assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}" + raise_agentic_failure( + GuardrailAssertionError, + f"Guardrail assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py index 7b29bf624..01d4ae04f 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py @@ -18,6 +18,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -37,7 +38,6 @@ LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -528,20 +528,21 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.evaluation - detail = { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "disambiguated": ev.disambiguated, - "actual_create_args": best.actual_create_args, - "actual_execute_result": best.actual_execute_result, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + disambiguated=ev.disambiguated, + actual_create_args=best.actual_create_args, + actual_execute_result=best.actual_execute_result, # Why the loop stopped -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -552,21 +553,23 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Actual create args: {best.actual_create_args}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = KdaSkillAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + raise_agentic_failure( + KdaSkillAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, + ) + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py index e2f95109e..a4d3c9e88 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py @@ -19,6 +19,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -39,7 +40,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings @@ -549,44 +549,45 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best expected_outputs_list: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output] - detail = { - "metric_created": best.metric_created, - "maql_correct": best.maql_correct, - "expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list], - "actual_maql": best.actual_maql, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + metric_created=best.metric_created, + maql_correct=best.maql_correct, + expected_maql_candidates=[c.get("maql", "") for c in expected_outputs_list], + actual_maql=best.actual_maql, # Why the loop stopped. metric_created=False alone cannot tell a refusal from a run # that hit max_iterations while still on track -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) candidates_str = "; ".join(repr(c.get("maql", "")) for c in expected_outputs_list) - exc = MetricSkillAssertionError( + raise_agentic_failure( + MetricSkillAssertionError, f"Metric skill assertion failed. {gate_note} " f"metric_created={best.metric_created}, maql_correct={best.maql_correct}. " f"Expected MAQL (candidates): {candidates_str}. " - f"Actual MAQL: {best.actual_maql}." + f"Actual MAQL: {best.actual_maql}.", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.best_run_latency_s = run_latency_s(best.timings) - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, - timings=item_timings, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py index 95fd9ef8a..a11c3cf24 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py @@ -14,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -30,7 +31,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, ) _DEFAULT_K = 1 @@ -263,34 +263,35 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail = { - "tool_selected": best.tool_selected, - "tool_correct": best.tool_correct, - "tool_call_names": best.tool_call_names, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + tool_selected=best.tool_selected, + tool_correct=best.tool_correct, + tool_call_names=best.tool_call_names, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) - exc = SearchToolAssertionError( + raise_agentic_failure( + SearchToolAssertionError, f"Search tool assertion failed. {gate_note} " f"tool_selected={best.tool_selected}, tool_correct={best.tool_correct}. " - f"Tool calls made: {best.tool_call_names}" + f"Tool calls made: {best.tool_call_names}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py index 9ad64b722..382b12877 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py @@ -19,6 +19,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -46,7 +47,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.scoring import get_dimension_uri_set, get_metric_uri_set, uri_to_display_name @@ -485,15 +485,16 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.eval_result - detail = { + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, **evaluation_result_detail(ev), # Why the loop stopped -- see LoopExit. total_turns is already the turn count for # this run, so it doubles as turns_used. - "exit_reason": best.exit_reason.value, - "turns_used": int(best.total_turns), - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=int(best.total_turns), + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -502,7 +503,8 @@ def _write_scores(ctx: RunTraceContext) -> None: cross_ref_detail = (" → " + "; ".join(ev.cross_ref_errors)) if ev.cross_ref_errors else "" expected_dump = best.best_expected.model_dump(exclude_none=True) actual_dump = best.actual_output.model_dump(exclude_none=True) if best.actual_output else None - exc = VisualizationAssertionError( + raise_agentic_failure( + VisualizationAssertionError, "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" "Agentic Visualization Assertion Failed! (Critical Mode)\n" f"{gate_note}\n" @@ -530,22 +532,21 @@ def _write_scores(ctx: RunTraceContext) -> None: f" attribute : {ev.filter_attribute_score}\n" f"{_filter_diff('attribute', ev)}" f" Viz Type Hard : {ev.viz_type_hard}\n" - "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" + "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py index d51ac6fde..255f369dc 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py @@ -26,6 +26,7 @@ from dataclasses import dataclass, field from typing import Any +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -43,9 +44,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -206,6 +207,10 @@ class WhatIfRunResult: # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from # turn_wall_clock_sec above, which is only the scenario-execution turn. run_latency_s: float = 0.0 + # Why the simulated-user loop stopped -- see LoopExit. `triggered=False` alone cannot + # separate a refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -340,6 +345,10 @@ def _accumulate(result: ChatResult) -> None: all_tool_call_events.extend(result.tool_call_events or []) all_reasoning_step_events.extend(result.reasoning_step_events or []) + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED + for iteration in range(max_iterations): try: chat_result = client.send_message(conv_id, current_question) @@ -352,6 +361,7 @@ def _accumulate(result: ChatResult) -> None: _accumulate(partial) create_args, execute_result = _extract_what_if_calls(all_tool_call_events) turn_completed = False + exit_reason = LoopExit.CHAT_ERROR break reasoning_steps.extend(chat_result.reasoning_steps or []) response_id = chat_result.response_id or response_id @@ -365,8 +375,10 @@ def _accumulate(result: ChatResult) -> None: if execute_result is not None: # The turn that ran the scenario, not an earlier disambiguation turn. turn_wall_clock_sec = chat_result.turn_wall_clock_sec + exit_reason = LoopExit.SUCCESS break if not response_text: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -375,6 +387,7 @@ def _accumulate(result: ChatResult) -> None: disambiguated = True except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run _log.warning("Simulated what-if user reply failed for conversation %s: %s", conv_id, exc) + exit_reason = LoopExit.SIMULATED_USER_FAILED break return WhatIfRunResult( @@ -382,6 +395,7 @@ def _accumulate(result: ChatResult) -> None: evaluation=_evaluate_run(create_args, execute_result, expected_output, turn_completed, disambiguated), actual_create_args=create_args, actual_execute_result=execute_result, + exit_reason=exit_reason, turn_wall_clock_sec=turn_wall_clock_sec, reasoning_steps=reasoning_steps, response_id=response_id, @@ -439,26 +453,30 @@ class WhatIfAssertionError(AgenticAssertionError): def _detail(best: WhatIfRunResult) -> dict[str, Any]: ev = best.evaluation adjustments = _adjustments(best.actual_create_args) - return { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "metric_correct": ev.metric_correct, - "maql_correct": ev.maql_correct, - "scenario_count_correct": ev.scenario_count_correct, - "baseline_correct": ev.baseline_correct, + return agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + metric_correct=ev.metric_correct, + maql_correct=ev.maql_correct, + scenario_count_correct=ev.scenario_count_correct, + baseline_correct=ev.baseline_correct, # Which content checks the fixture pinned -- without it a run that verified nothing # reads the same as one where everything matched. - "asserted": ev.asserted, - "disambiguated": ev.disambiguated, - "actual_adjustments": adjustments, - "actual_scenario_labels": [ + asserted=ev.asserted, + disambiguated=ev.disambiguated, + actual_adjustments=adjustments, + actual_scenario_labels=[ s.get("label") for s in ((best.actual_create_args or {}).get("scenarios") or []) if isinstance(s, dict) ], - "actual_execute_result": best.actual_execute_result, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + actual_execute_result=best.actual_execute_result, + # Why the loop stopped. triggered=False alone cannot tell a refusal from a run + # that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) def evaluate_agentic_what_if( @@ -581,22 +599,24 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Actual adjustments: {detail['actual_adjustments']}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = WhatIfAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = len(summary.run_results) - exc.best_run_latency_s = best.run_latency_s - raise exc - - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=len(summary.run_results), + raise_agentic_failure( + WhatIfAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, + ) + + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py b/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py index 84e138fb1..f3aa69534 100644 --- a/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py +++ b/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py @@ -16,7 +16,7 @@ evaluate_agentic_anomaly_detection, run_agentic_anomaly_detection, ) -from gooddata_eval.core.models import ChatResult +from gooddata_eval.core.models import ChatResult, LoopExit _MODULE = "gooddata_eval.core.agentic.anomaly_detection" @@ -254,18 +254,22 @@ def test_a_single_turn_detection_passes(): summary = _run([_chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is False + assert summary.best.exit_reason is LoopExit.SUCCESS def test_a_clarifying_question_is_answered_and_the_run_continues(): summary = _run([_chat([], text="Which Spend metric did you mean?"), _chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is True + assert summary.best.exit_reason is LoopExit.SUCCESS def test_the_loop_stops_at_max_iterations_without_a_detection(): + # executed=False alone cannot tell this apart from a refusal -- exit_reason can. summary = _run([_chat([], text="Still thinking")] * 4, max_iterations=4) assert summary.pass_at_k is False assert summary.best.evaluation.executed is False + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_response_ends_the_run_immediately(): @@ -276,17 +280,19 @@ def test_an_empty_response_ends_the_run_immediately(): patch(f"{_MODULE}.ChatClient", return_value=client), patch(f"{_MODULE}.generate_simulated_anomaly_response") as sim, ): - run_agentic_anomaly_detection( + summary = run_agentic_anomaly_detection( host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED ) assert client.send_message.call_count == 1 sim.assert_not_called() + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_chat_error_on_a_later_run_does_not_discard_the_earlier_one(): summary = _run([_chat(_pair()), RuntimeError("boom")], k=2) assert len(summary.run_results) == 2 assert summary.pass_at_k is True + assert summary.run_results[1].exit_reason is LoopExit.CHAT_ERROR def test_k_must_be_at_least_one(): @@ -318,6 +324,8 @@ def test_detail_reports_the_series_analysed_and_the_flag_count(): assert outcome.detail["anomaly_point_count"] == 0 assert outcome.detail["asserted"] == ["metric", "granularity"] assert "latency_breakdown" in outcome.detail + assert "tool_calls" in outcome.detail + assert outcome.detail["exit_reason"] == "success" assert outcome.runs_passed == 1 diff --git a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py index e949d8a3b..49620a8e8 100644 --- a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py @@ -15,7 +15,7 @@ evaluate_dashboard_response, run_agentic_dashboard_skill, ) -from gooddata_eval.core.models import ChatResult, ToolCallEvent +from gooddata_eval.core.models import ChatResult, LoopExit, ToolCallEvent # Ids and titles are the ones the eval layout seeds; shapes are trimmed from real runs of # `agent_dashboard_skill` against ecommerce_demo. @@ -1086,6 +1086,8 @@ def test_the_loop_is_capped(self): assert client.send_message.call_count == 2 assert not summary.pass_at_k assert not summary.best.evaluation.drafted + # drafted=False alone cannot tell this apart from a refusal -- exit_reason can. + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_turn_stops_the_loop(self): client = MagicMock() @@ -1093,6 +1095,7 @@ def test_an_empty_turn_stops_the_loop(self): summary = _run_with(client, _DC05_EXPECTED, max_iterations=5) assert client.send_message.call_count == 1 assert not summary.pass_at_k + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_caller_supplied_conversation_is_not_deleted(self): client = MagicMock() @@ -1161,6 +1164,8 @@ def test_a_passing_run_returns_the_outcome(self): assert outcome.runs_passed == 1 assert outcome.runs_effective == 1 assert outcome.detail["charts_matched"] is True + assert outcome.detail["exit_reason"] == "success" + assert "tool_calls" in outcome.detail def _as_events(raw: list[dict]) -> list[ToolCallEvent]: diff --git a/packages/gooddata-eval/tests/test_agentic_general_question.py b/packages/gooddata-eval/tests/test_agentic_general_question.py index cd9db187e..26dddbacc 100644 --- a/packages/gooddata-eval/tests/test_agentic_general_question.py +++ b/packages/gooddata-eval/tests/test_agentic_general_question.py @@ -257,6 +257,7 @@ def test_evaluate_agentic_general_question_returns_reasoning_steps_on_pass(): "judge_reasoning": "Correct answer", "actual_output": "42", "latency_breakdown": [], + "tool_calls": [], } @@ -284,6 +285,7 @@ def test_evaluate_agentic_general_question_attaches_reasoning_steps_to_exception "judge_reasoning": "Wrong answer", "actual_output": "I don't know", "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_kda_skill.py b/packages/gooddata-eval/tests/test_agentic_kda_skill.py index 573a2a267..3d6f5b620 100644 --- a/packages/gooddata-eval/tests/test_agentic_kda_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_kda_skill.py @@ -1134,6 +1134,7 @@ def test_evaluate_agentic_kda_skill_returns_reasoning_steps_on_pass(): "turns_used": 1, "max_iterations": 1, "latency_breakdown": [], + "tool_calls": [], } @@ -1172,6 +1173,7 @@ def test_evaluate_agentic_kda_skill_attaches_reasoning_steps_to_exception_on_fai "turns_used": 1, "max_iterations": 1, "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_search_tool.py b/packages/gooddata-eval/tests/test_agentic_search_tool.py index 975004cd4..f1100d9c9 100644 --- a/packages/gooddata-eval/tests/test_agentic_search_tool.py +++ b/packages/gooddata-eval/tests/test_agentic_search_tool.py @@ -208,6 +208,7 @@ def test_evaluate_agentic_search_tool_returns_reasoning_steps_on_pass(): "tool_correct": True, "tool_call_names": ["search_objects"], "latency_breakdown": [], + "tool_calls": [], } @@ -244,4 +245,5 @@ def test_evaluate_agentic_search_tool_attaches_reasoning_steps_to_exception_on_f "tool_correct": False, "tool_call_names": [], "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_what_if.py b/packages/gooddata-eval/tests/test_agentic_what_if.py index 07da79ca8..baf267442 100644 --- a/packages/gooddata-eval/tests/test_agentic_what_if.py +++ b/packages/gooddata-eval/tests/test_agentic_what_if.py @@ -13,7 +13,7 @@ evaluate_agentic_what_if, run_agentic_what_if, ) -from gooddata_eval.core.models import ChatResult +from gooddata_eval.core.models import ChatResult, LoopExit _MODULE = "gooddata_eval.core.agentic.what_if" @@ -215,6 +215,7 @@ def test_a_single_turn_scenario_passes(): summary = _run([_chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is False + assert summary.best.exit_reason is LoopExit.SUCCESS def test_a_clarifying_question_is_answered_and_the_run_continues(): @@ -222,12 +223,15 @@ def test_a_clarifying_question_is_answered_and_the_run_continues(): summary = _run([_chat([], text="Which Spend metric did you mean?"), _chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is True + assert summary.best.exit_reason is LoopExit.SUCCESS def test_the_loop_stops_at_max_iterations_without_a_scenario(): + # triggered=False alone cannot tell this apart from a refusal -- exit_reason can. summary = _run([_chat([], text="Still thinking")] * 4, max_iterations=4) assert summary.pass_at_k is False assert summary.best.evaluation.triggered is False + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_response_ends_the_run_immediately(): @@ -238,15 +242,19 @@ def test_an_empty_response_ends_the_run_immediately(): patch(f"{_MODULE}.ChatClient", return_value=client), patch(f"{_MODULE}.generate_simulated_what_if_response") as sim, ): - run_agentic_what_if(host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED) + summary = run_agentic_what_if( + host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED + ) assert client.send_message.call_count == 1 sim.assert_not_called() + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_chat_error_on_a_later_run_does_not_discard_the_earlier_one(): summary = _run([_chat(_pair()), RuntimeError("boom")], k=2) assert len(summary.run_results) == 2 assert summary.pass_at_k is True + assert summary.run_results[1].exit_reason is LoopExit.CHAT_ERROR def test_k_must_be_at_least_one(): @@ -277,6 +285,8 @@ def test_detail_reports_the_adjustments_the_agent_asked_for(): assert outcome.detail["actual_scenario_labels"] == ["Scenario A"] assert outcome.detail["asserted"] == ["metric_id", "scenario_maql"] assert "latency_breakdown" in outcome.detail + assert "tool_calls" in outcome.detail + assert outcome.detail["exit_reason"] == "success" assert outcome.runs_passed == 1 From b6d35c512a99bc510d6c7f3fdb6974bb2cf2827d Mon Sep 17 00:00:00 2001 From: Peter Tomko Date: Wed, 7 Oct 2026 23:14:39 +0200 Subject: [PATCH 4/4] fix(gooddata-eval): record a dashboard chat fault instead of letting it escape ChatClient.send_message can raise ChatError, and the dashboard loop was the only one that let it through. Escaping run_agentic_dashboard_skill discarded every K-run already completed along with this one's exit_reason, so the item surfaced as a bare error with no per-run diagnosis -- exactly what the shared LoopExit contract exists to prevent. metric_skill.py and conversation.py already record it; this matches them. A bare catch would have discarded ChatError.partial_result. What the stream delivered before it died is still the turn's work, so its reasoning, tool calls and step count are merged -- shifted and indexed the way a completed turn is, since left raw a late turn's call_ts restarts near zero and timeline_detail would report it as overlapping the first. DashboardRunResult's comment claimed CHAT_ERROR was unreachable here because the loop had no try/except. That had it backwards: the missing try/except was the bug, not a reason the fault could not occur. SIMULATED_USER_FAILED really is unreachable -- the follow-up is built by build_simulated_reply with no judge call to fail -- and the comment now says so. Three tests, each verified to fail without the fix. Co-Authored-By: Claude Opus 5 --- .../core/agentic/dashboard_skill.py | 35 +++++++++++-- .../tests/test_agentic_dashboard_skill.py | 49 +++++++++++++++++++ 2 files changed, 80 insertions(+), 4 deletions(-) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py index a29b141d7..02e5483b2 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py @@ -28,7 +28,7 @@ utc_now, ) from gooddata_eval.core.chat.render import render_answer_text -from gooddata_eval.core.chat.sse_client import ChatClient +from gooddata_eval.core.chat.sse_client import ChatClient, ChatError from gooddata_eval.core.config import ReasoningEffort from gooddata_eval.core.models import ( AgenticAssertionError, @@ -667,8 +667,9 @@ class DashboardRunResult: timings: PhaseTimings = field(default_factory=PhaseTimings) # Why the loop stopped -- see LoopExit. A dashboard not produced alone cannot tell a # refusal from a run that hit max_iterations while still on track (matches - # kda_skill.py/alert_skill.py). No CHAT_ERROR/SIMULATED_USER_FAILED member here: this - # loop has neither a send_message try/except nor a simulated-user call to fail. + # kda_skill.py/alert_skill.py). CHAT_ERROR is reachable; SIMULATED_USER_FAILED is not -- + # this loop's follow-up is built from the expectation by build_simulated_reply, with no + # judge call that could fail. exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @@ -879,7 +880,33 @@ def _execute_single_dashboard_run( for iteration in range(max_iterations): turns += 1 agent_started = time.monotonic() - chat_result = client.send_message(conversation_id, current_question) + try: + chat_result = client.send_message(conversation_id, current_question) + except ChatError as exc: + # Recorded rather than left to escape: this runs inside one of K runs, so an + # uncaught ChatError discarded every run already completed along with this one's + # exit_reason, and the item surfaced as a bare error with no per-run diagnosis. + # Matches metric_skill.py and conversation.py, which record it the same way. + timings.agent_s += time.monotonic() - agent_started + print(f"[CHAT] send_message failed for conversation {conversation_id}: {exc}") + partial = exc.partial_result + if partial is not None: + # Shifted and indexed exactly as a completed turn is, before anything reads + # the events: left raw, a late turn's call_ts restarts near zero and + # timeline_detail reports it as overlapping the first one. + reasoning_steps.extend(partial.reasoning_steps or []) + response_id = partial.response_id or response_id + turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events( + partial, + turn_offset=turn_offset, + tool_index_offset=tool_index_offset, + reasoning_index_offset=reasoning_index_offset, + ) + all_tool_call_events.extend(partial.tool_call_events or []) + all_reasoning_step_events.extend(partial.reasoning_step_events or []) + steps += partial.reasoning_step_count + exit_reason = LoopExit.CHAT_ERROR + break agent_elapsed = time.monotonic() - agent_started timings.agent_s += agent_elapsed reasoning_steps.extend(chat_result.reasoning_steps or []) diff --git a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py index 49620a8e8..23b26a7b2 100644 --- a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py @@ -15,6 +15,7 @@ evaluate_dashboard_response, run_agentic_dashboard_skill, ) +from gooddata_eval.core.chat.sse_client import ChatError from gooddata_eval.core.models import ChatResult, LoopExit, ToolCallEvent # Ids and titles are the ones the eval layout seeds; shapes are trimmed from real runs of @@ -1104,6 +1105,54 @@ def test_a_caller_supplied_conversation_is_not_deleted(self): client.create_conversation.assert_not_called() client.delete_conversation.assert_not_called() + def test_a_chat_error_is_recorded_rather_than_left_to_escape(self): + """It used to escape run_agentic_dashboard_skill, taking every completed K-run with + it: no DashboardRunResult, no exit_reason, just a bare error on the item. Recorded + the way metric_skill.py and conversation.py record it, so an infrastructure fault + stays distinguishable from the agent failing to draft.""" + client = MagicMock() + client.send_message.side_effect = ChatError("gen-ai fell over") + + summary = _run_with(client, _DC05_EXPECTED) + + assert summary.run_results[0].exit_reason is LoopExit.CHAT_ERROR + assert summary.run_results[0].evaluation.strict_pass is False + assert summary.pass_at_k is False + + def test_a_chat_error_keeps_the_partial_turn_s_work(self): + """A bare catch would discard ChatError.partial_result. What the stream delivered + before it died is still this turn's work -- the steps were taken and the tools ran.""" + partial = _chat_result( + tool_calls=[{"functionName": "search_objects", "functionArguments": "{}", "callTs": 1, "durationMs": 10}], + text="Here is the draft.", + ) + partial.reasoning_steps = ["looked for the charts"] + partial.reasoning_step_count = 1 + client = MagicMock() + client.send_message.side_effect = ChatError("stream died mid-answer", partial_result=partial) + + summary = _run_with(client, _DC05_EXPECTED) + run = summary.run_results[0] + + assert run.exit_reason is LoopExit.CHAT_ERROR + assert run.reasoning_steps == ["looked for the charts"] + assert run.total_steps == 1 + assert [tc.function_name for tc in run.tool_call_events] == ["search_objects"] + + def test_a_chat_error_on_a_later_turn_keeps_the_earlier_ones(self): + """The whole point of recording it: turns already completed survive the fault.""" + client = MagicMock() + client.send_message.side_effect = [ + _chat_result(text="Which charts did you have in mind?"), + ChatError("gen-ai fell over"), + ] + + summary = _run_with(client, _DC05_EXPECTED) + run = summary.run_results[0] + + assert run.exit_reason is LoopExit.CHAT_ERROR + assert run.total_turns == 2 + class TestEvaluateEntryPoint: def test_a_failing_run_raises_with_the_failures_attached(self):