Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -369,6 +369,18 @@ def _apply_timings(item_report: ItemReport, timings: Any) -> None:
item_report.simulated_user_latency_s = timings.simulated_user_s


def _apply_best_run_latency(item_report: ItemReport, source: Any) -> None:
"""Copy the rank-selected run's own wall time, mirroring the single-shot path's
``ItemReport.best_run_latency_s`` (see ``core/runner.py``'s ``_run_one_item``).

Kinds not yet wired to measure it pass None and keep the field at its own None
default rather than reporting an invented number.
"""
best_run_latency_s = getattr(source, "best_run_latency_s", None)
if best_run_latency_s is not None:
item_report.best_run_latency_s = best_run_latency_s


def run_agentic_items(
items: list[DatasetItem],
host: str,
Expand Down Expand Up @@ -454,6 +466,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport:
item_report.response_id = response_id
item_report.best_detail = detail or {}
_apply_timings(item_report, getattr(outcome, "timings", None))
_apply_best_run_latency(item_report, outcome)
_apply_run_counts(item_report, outcome)
except AssertionError as exc:
item_report.gate_passed = False if gated else None
Expand All @@ -463,6 +476,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport:
item_report.response_id = getattr(exc, "response_id", None)
item_report.best_detail = getattr(exc, "detail", None) or {}
_apply_timings(item_report, getattr(exc, "timings", None))
_apply_best_run_latency(item_report, exc)
_apply_run_counts(item_report, exc)
# Read off the counts, not off the gate: pass^K fails items where runs did pass,
# and reporting those as pass_at_k False would contradict the Langfuse score of
Expand All @@ -477,6 +491,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport:
# judge broke should not also report the agent as costing 0s. Kinds that
# attach no timings to the exception keep their 0.0 defaults.
_apply_timings(item_report, getattr(exc, "timings", None))
_apply_best_run_latency(item_report, exc)
finally:
item_report.latency_s = time.perf_counter() - t0

Expand Down
109 changes: 109 additions & 0 deletions packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,109 @@
# (C) 2026 GoodData Corporation
"""The common tail of every ``evaluate_agentic_*`` function.

Before this module, each of the agentic evaluators hand-wrote the same block: copy
``reasoning_steps``/``conversation_id``/``response_id``/``detail``/``runs_passed``/
``runs_effective``/``best_run_latency_s`` (and sometimes ``timings``) either onto a
raised exception or into the returned ``AgenticEvalOutcome``. Eleven independent copies
of the same ~7 lines meant a new universal field needed eleven edits, not one, and
nothing failed loudly when a copy was missed -- exactly what happened to ``timings``
(see ``AgenticAssertionError``'s docstring: "while it lived in eight copies, two
declared timings and six did not") and again to ``best_run_latency_s`` (every agentic
kind silently reported it as ``null`` until this was noticed and fixed kind-by-kind).

``agentic_detail`` has the same motivation for the ``detail`` dict's timeline fields:
``timeline_detail`` builds both ``latency_breakdown`` and ``tool_calls`` from the same
events so they stay index-aligned, but roughly half the evaluators called
``build_latency_breakdown`` directly and silently never got a ``tool_calls`` key.
"""

from __future__ import annotations

from typing import Any, NoReturn

from gooddata_eval.core.models import (
AgenticAssertionError,
AgenticEvalOutcome,
ReasoningStepEvent,
ToolCallEvent,
timeline_detail,
)
from gooddata_eval.core.timing import PhaseTimings

__all__ = ["agentic_detail", "agentic_success", "raise_agentic_failure"]


def agentic_detail(
tool_call_events: list[ToolCallEvent],
reasoning_step_events: list[ReasoningStepEvent] | None,
**kind_specific: Any,
) -> dict:
"""A kind's full ``detail`` dict: its own fields plus the universal timeline ones.

``kind_specific`` comes first in the merge so a kind can never accidentally shadow
``latency_breakdown``/``tool_calls`` with a same-named field of its own.
"""
return {**kind_specific, **timeline_detail(tool_call_events, reasoning_step_events)}


def raise_agentic_failure(
exception_cls: type[AgenticAssertionError],
message: str,
*,
reasoning_steps: list[str],
conversation_id: str,
response_id: str | None,
detail: dict,
runs_passed: int,
runs_effective: int,
best_run_latency_s: float | None,
timings: PhaseTimings | None = None,
) -> NoReturn:
"""Build ``exception_cls(message)`` with every common field attached, and raise it.

``exception_cls`` must be an ``AgenticAssertionError`` subclass -- that base class is
what declares these fields as legal targets (see its docstring). A kind whose own
"no verdict at all" branch raises a bare ``JudgeResponseError`` instead (not a subclass)
does not go through this helper for that branch.

``timings`` stays optional: only the three kinds that track per-run ``PhaseTimings``
(general_question, metric_skill, dashboard_skill) pass one -- the rest keep their
own ``PhaseTimings()`` zero default, same as before this helper existed.
"""
exc = exception_cls(message)
exc.reasoning_steps = reasoning_steps
exc.conversation_id = conversation_id
exc.response_id = response_id
exc.detail = detail
exc.runs_passed = runs_passed
exc.runs_effective = runs_effective
exc.best_run_latency_s = best_run_latency_s
if timings is not None:
exc.timings = timings
raise exc


def agentic_success(
*,
reasoning_steps: list[str],
conversation_id: str | None,
response_id: str | None,
detail: dict,
runs_passed: int,
runs_effective: int,
best_run_latency_s: float | None,
timings: PhaseTimings | None = None,
) -> AgenticEvalOutcome:
"""The success-path mirror of ``raise_agentic_failure`` -- same fields, same shape."""
kwargs: dict[str, Any] = {
"reasoning_steps": reasoning_steps,
"conversation_id": conversation_id,
"response_id": response_id,
"detail": detail,
"runs_passed": runs_passed,
"runs_effective": runs_effective,
"best_run_latency_s": best_run_latency_s,
}
if timings is not None:
kwargs["timings"] = timings
return AgenticEvalOutcome(**kwargs)
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
import json
import os
import re
import time
from dataclasses import dataclass, field
from typing import Any

Expand All @@ -20,6 +21,7 @@
log_gate_scores,
stamp_gate_metadata,
)
from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure
from gooddata_eval.core.agentic._trace_linker import (
RunIdentity,
RunTraceContext,
Expand All @@ -39,7 +41,6 @@
ReasoningStepEvent,
ToolCallEvent,
shift_and_index_events,
timeline_detail,
)

try:
Expand Down Expand Up @@ -490,6 +491,9 @@ class AlertRunResult:
# also report operator/threshold/metric/recipients as False.
exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED
turns_used: int = 0
# This run's own wall time, across all its turns -- mirrors the single-shot
# path's best_run_latency_s (see core/runner.py's _run_one_item).
run_latency_s: float = 0.0


@dataclass
Expand Down Expand Up @@ -676,6 +680,7 @@ def run_agentic_alert_skill(

def _run_once(conv_id: str) -> AlertRunResult:
alert_id_to_delete: str | None = None
run_started = time.monotonic()
try:
alert_id: str | None = None
actual_args: dict = {}
Expand Down Expand Up @@ -794,6 +799,7 @@ def _run_once(conv_id: str) -> AlertRunResult:
reasoning_step_events=all_reasoning_step_events,
exit_reason=exit_reason,
turns_used=turns_used,
run_latency_s=time.monotonic() - run_started,
)
finally:
if alert_id_to_delete:
Expand Down Expand Up @@ -953,49 +959,52 @@ def _write_scores(ctx: RunTraceContext) -> None:

best = summary.best
ev = best.eval
detail = {
"alert_created": ev.alert_created,
"operator_correct": ev.operator_correct,
"threshold_correct": ev.threshold_correct,
"trigger_correct": ev.trigger_correct,
"filters_correct": ev.filters_correct,
"metric_correct": ev.metric_correct,
"recipients_correct": ev.recipients_correct,
"attributes_correct": ev.attributes_correct,
"granularity_correct": ev.granularity_correct,
"actual_alert_arguments": best.actual_alert_arguments,
detail = agentic_detail(
best.tool_call_events,
best.reasoning_step_events,
alert_created=ev.alert_created,
operator_correct=ev.operator_correct,
threshold_correct=ev.threshold_correct,
trigger_correct=ev.trigger_correct,
filters_correct=ev.filters_correct,
metric_correct=ev.metric_correct,
recipients_correct=ev.recipients_correct,
attributes_correct=ev.attributes_correct,
granularity_correct=ev.granularity_correct,
actual_alert_arguments=best.actual_alert_arguments,
# Why the loop stopped. alert_created=False alone cannot tell a refusal from a run
# that hit max_iterations while still on track -- see LoopExit.
"exit_reason": best.exit_reason.value,
"turns_used": best.turns_used,
"max_iterations": max_iterations,
**timeline_detail(best.tool_call_events, best.reasoning_step_events),
}
exit_reason=best.exit_reason.value,
turns_used=best.turns_used,
max_iterations=max_iterations,
)

if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k):
gate_note = gate_failure_note(gate, runs_passed, runs_effective)
exc = AlertSkillAssertionError(
raise_agentic_failure(
AlertSkillAssertionError,
f"Alert skill assertion failed. {gate_note} strict_pass={ev.strict_pass}. "
f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
f"filters_correct={ev.filters_correct}, metric_correct={ev.metric_correct}, "
f"recipients_correct={ev.recipients_correct}, "
f"attributes_correct={ev.attributes_correct}, "
f"granularity_correct={ev.granularity_correct}. "
f"Actual args: {best.actual_alert_arguments}"
f"Actual args: {best.actual_alert_arguments}",
reasoning_steps=best.reasoning_steps,
conversation_id=best.conversation_id,
response_id=best.response_id,
detail=detail,
runs_passed=runs_passed,
runs_effective=runs_effective,
best_run_latency_s=best.run_latency_s,
)
exc.reasoning_steps = best.reasoning_steps
exc.conversation_id = best.conversation_id
exc.response_id = best.response_id
exc.detail = detail
exc.runs_passed = runs_passed
exc.runs_effective = runs_effective
raise exc
return AgenticEvalOutcome(
runs_passed=runs_passed,
runs_effective=runs_effective,
return agentic_success(
reasoning_steps=best.reasoning_steps,
conversation_id=best.conversation_id,
response_id=best.response_id,
detail=detail,
runs_passed=runs_passed,
runs_effective=runs_effective,
best_run_latency_s=best.run_latency_s,
)
Loading
Loading