Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
109 changes: 109 additions & 0 deletions packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,109 @@
# (C) 2026 GoodData Corporation
"""The common tail of every ``evaluate_agentic_*`` function.

Before this module, each of the agentic evaluators hand-wrote the same block: copy
``reasoning_steps``/``conversation_id``/``response_id``/``detail``/``runs_passed``/
``runs_effective``/``best_run_latency_s`` (and sometimes ``timings``) either onto a
raised exception or into the returned ``AgenticEvalOutcome``. Eleven independent copies
of the same ~7 lines meant a new universal field needed eleven edits, not one, and
nothing failed loudly when a copy was missed -- exactly what happened to ``timings``
(see ``AgenticAssertionError``'s docstring: "while it lived in eight copies, two
declared timings and six did not") and again to ``best_run_latency_s`` (every agentic
kind silently reported it as ``null`` until this was noticed and fixed kind-by-kind).

``agentic_detail`` has the same motivation for the ``detail`` dict's timeline fields:
``timeline_detail`` builds both ``latency_breakdown`` and ``tool_calls`` from the same
events so they stay index-aligned, but roughly half the evaluators called
``build_latency_breakdown`` directly and silently never got a ``tool_calls`` key.
"""

from __future__ import annotations

from typing import Any, NoReturn

from gooddata_eval.core.models import (
AgenticAssertionError,
AgenticEvalOutcome,
ReasoningStepEvent,
ToolCallEvent,
timeline_detail,
)
from gooddata_eval.core.timing import PhaseTimings

__all__ = ["agentic_detail", "agentic_success", "raise_agentic_failure"]


def agentic_detail(
tool_call_events: list[ToolCallEvent],
reasoning_step_events: list[ReasoningStepEvent] | None,
**kind_specific: Any,
) -> dict:
"""A kind's full ``detail`` dict: its own fields plus the universal timeline ones.

``kind_specific`` comes first in the merge so a kind can never accidentally shadow
``latency_breakdown``/``tool_calls`` with a same-named field of its own.
"""
return {**kind_specific, **timeline_detail(tool_call_events, reasoning_step_events)}


def raise_agentic_failure(
exception_cls: type[AgenticAssertionError],
message: str,
*,
reasoning_steps: list[str],
conversation_id: str,
response_id: str | None,
detail: dict,
runs_passed: int,
runs_effective: int,
best_run_latency_s: float | None,
timings: PhaseTimings | None = None,
) -> NoReturn:
"""Build ``exception_cls(message)`` with every common field attached, and raise it.

``exception_cls`` must be an ``AgenticAssertionError`` subclass -- that base class is
what declares these fields as legal targets (see its docstring). A kind whose own
"no verdict at all" branch raises a bare ``JudgeResponseError`` instead (not a subclass)
does not go through this helper for that branch.

``timings`` stays optional: only the three kinds that track per-run ``PhaseTimings``
(general_question, metric_skill, dashboard_skill) pass one -- the rest keep their
own ``PhaseTimings()`` zero default, same as before this helper existed.
"""
exc = exception_cls(message)
exc.reasoning_steps = reasoning_steps
exc.conversation_id = conversation_id
exc.response_id = response_id
exc.detail = detail
exc.runs_passed = runs_passed
exc.runs_effective = runs_effective
exc.best_run_latency_s = best_run_latency_s
if timings is not None:
exc.timings = timings
raise exc


def agentic_success(
*,
reasoning_steps: list[str],
conversation_id: str | None,
response_id: str | None,
detail: dict,
runs_passed: int,
runs_effective: int,
best_run_latency_s: float | None,
timings: PhaseTimings | None = None,
) -> AgenticEvalOutcome:
"""The success-path mirror of ``raise_agentic_failure`` -- same fields, same shape."""
kwargs: dict[str, Any] = {
"reasoning_steps": reasoning_steps,
"conversation_id": conversation_id,
"response_id": response_id,
"detail": detail,
"runs_passed": runs_passed,
"runs_effective": runs_effective,
"best_run_latency_s": best_run_latency_s,
}
if timings is not None:
kwargs["timings"] = timings
return AgenticEvalOutcome(**kwargs)
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@
log_gate_scores,
stamp_gate_metadata,
)
from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure
from gooddata_eval.core.agentic._trace_linker import (
RunIdentity,
RunTraceContext,
Expand All @@ -40,7 +41,6 @@
ReasoningStepEvent,
ToolCallEvent,
shift_and_index_events,
timeline_detail,
)

try:
Expand Down Expand Up @@ -951,51 +951,52 @@ def _write_scores(ctx: RunTraceContext) -> None:

best = summary.best
ev = best.eval
detail = {
"alert_created": ev.alert_created,
"operator_correct": ev.operator_correct,
"threshold_correct": ev.threshold_correct,
"trigger_correct": ev.trigger_correct,
"filters_correct": ev.filters_correct,
"metric_correct": ev.metric_correct,
"recipients_correct": ev.recipients_correct,
"attributes_correct": ev.attributes_correct,
"granularity_correct": ev.granularity_correct,
"actual_alert_arguments": best.actual_alert_arguments,
detail = agentic_detail(
best.tool_call_events,
best.reasoning_step_events,
alert_created=ev.alert_created,
operator_correct=ev.operator_correct,
threshold_correct=ev.threshold_correct,
trigger_correct=ev.trigger_correct,
filters_correct=ev.filters_correct,
metric_correct=ev.metric_correct,
recipients_correct=ev.recipients_correct,
attributes_correct=ev.attributes_correct,
granularity_correct=ev.granularity_correct,
actual_alert_arguments=best.actual_alert_arguments,
# Why the loop stopped. alert_created=False alone cannot tell a refusal from a run
# that hit max_iterations while still on track -- see LoopExit.
"exit_reason": best.exit_reason.value,
"turns_used": best.turns_used,
"max_iterations": max_iterations,
**timeline_detail(best.tool_call_events, best.reasoning_step_events),
}
exit_reason=best.exit_reason.value,
turns_used=best.turns_used,
max_iterations=max_iterations,
)

if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k):
gate_note = gate_failure_note(gate, runs_passed, runs_effective)
exc = AlertSkillAssertionError(
raise_agentic_failure(
AlertSkillAssertionError,
f"Alert skill assertion failed. {gate_note} strict_pass={ev.strict_pass}. "
f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
f"filters_correct={ev.filters_correct}, metric_correct={ev.metric_correct}, "
f"recipients_correct={ev.recipients_correct}, "
f"attributes_correct={ev.attributes_correct}, "
f"granularity_correct={ev.granularity_correct}. "
f"Actual args: {best.actual_alert_arguments}"
f"Actual args: {best.actual_alert_arguments}",
reasoning_steps=best.reasoning_steps,
conversation_id=best.conversation_id,
response_id=best.response_id,
detail=detail,
runs_passed=runs_passed,
runs_effective=runs_effective,
best_run_latency_s=best.run_latency_s,
)
exc.reasoning_steps = best.reasoning_steps
exc.conversation_id = best.conversation_id
exc.response_id = best.response_id
exc.detail = detail
exc.runs_passed = runs_passed
exc.runs_effective = runs_effective
exc.best_run_latency_s = best.run_latency_s
raise exc
return AgenticEvalOutcome(
runs_passed=runs_passed,
runs_effective=runs_effective,
return agentic_success(
reasoning_steps=best.reasoning_steps,
conversation_id=best.conversation_id,
response_id=best.response_id,
detail=detail,
runs_passed=runs_passed,
runs_effective=runs_effective,
best_run_latency_s=best.run_latency_s,
)
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
log_gate_scores,
stamp_gate_metadata,
)
from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure
from gooddata_eval.core.agentic._trace_linker import (
RunIdentity,
RunTraceContext,
Expand All @@ -56,9 +57,9 @@
AgenticAssertionError,
AgenticEvalOutcome,
ChatResult,
LoopExit,
ReasoningStepEvent,
ToolCallEvent,
build_latency_breakdown,
shift_and_index_events,
)

Expand Down Expand Up @@ -287,6 +288,10 @@ class AnomalyRunResult:
# path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from
# turn_wall_clock_sec above, which is only the final triggering turn.
run_latency_s: float = 0.0
# Why the simulated-user loop stopped -- see LoopExit. `triggered=False` alone cannot
# separate a refusal from a run that hit max_iterations while still on track (matches
# kda_skill.py/alert_skill.py).
exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED


@dataclass
Expand Down Expand Up @@ -391,6 +396,10 @@ def _accumulate(result: ChatResult) -> None:
all_tool_call_events.extend(result.tool_call_events or [])
all_reasoning_step_events.extend(result.reasoning_step_events or [])

# Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that
# simply runs out of range() is labelled correctly with no trailing else.
exit_reason = LoopExit.BUDGET_EXHAUSTED

for iteration in range(max_iterations):
try:
chat_result = client.send_message(conv_id, current_question)
Expand All @@ -403,6 +412,7 @@ def _accumulate(result: ChatResult) -> None:
_accumulate(partial)
viz_args, execute_result = _extract_anomaly_calls(all_tool_call_events)
turn_completed = False
exit_reason = LoopExit.CHAT_ERROR
break
reasoning_steps.extend(chat_result.reasoning_steps or [])
response_id = chat_result.response_id or response_id
Expand All @@ -416,8 +426,10 @@ def _accumulate(result: ChatResult) -> None:
if execute_result is not None:
# The turn that ran the detection, not an earlier disambiguation turn.
turn_wall_clock_sec = chat_result.turn_wall_clock_sec
exit_reason = LoopExit.SUCCESS
break
if not response_text:
exit_reason = LoopExit.AGENT_SILENT
break
if iteration >= max_iterations - 1:
break
Expand All @@ -426,6 +438,7 @@ def _accumulate(result: ChatResult) -> None:
disambiguated = True
except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run
_log.warning("Simulated anomaly user reply failed for conversation %s: %s", conv_id, exc)
exit_reason = LoopExit.SIMULATED_USER_FAILED
break

return AnomalyRunResult(
Expand All @@ -434,6 +447,7 @@ def _accumulate(result: ChatResult) -> None:
actual_visualization=viz_args,
actual_execute_result=execute_result,
turn_wall_clock_sec=turn_wall_clock_sec,
exit_reason=exit_reason,
reasoning_steps=reasoning_steps,
response_id=response_id,
tool_call_events=all_tool_call_events,
Expand Down Expand Up @@ -487,25 +501,29 @@ class AnomalyDetectionAssertionError(AgenticAssertionError):

def _detail(best: AnomalyRunResult) -> dict[str, Any]:
ev = best.evaluation
return {
"triggered": ev.triggered,
"executed": ev.executed,
"success": ev.success,
"turn_completed": ev.turn_completed,
"metric_correct": ev.metric_correct,
"granularity_correct": ev.granularity_correct,
return agentic_detail(
best.tool_call_events,
best.reasoning_step_events,
triggered=ev.triggered,
executed=ev.executed,
success=ev.success,
turn_completed=ev.turn_completed,
metric_correct=ev.metric_correct,
granularity_correct=ev.granularity_correct,
# Which content checks the fixture pinned -- without it a run that verified nothing
# reads the same as one where everything matched.
"asserted": ev.asserted,
"disambiguated": ev.disambiguated,
"actual_metrics": sorted(_metric_uris(best.actual_visualization)),
"actual_granularity": _inferred_granularity(best.actual_visualization),
asserted=ev.asserted,
disambiguated=ev.disambiguated,
actual_metrics=sorted(_metric_uris(best.actual_visualization)),
actual_granularity=_inferred_granularity(best.actual_visualization),
# Reported, never asserted: whether a real series contains anomalies is a property
# of the data, so a fixture demanding some would fail on the next warehouse refresh.
"anomaly_point_count": _point_count(best.actual_execute_result),
"actual_execute_result": best.actual_execute_result,
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
}
anomaly_point_count=_point_count(best.actual_execute_result),
actual_execute_result=best.actual_execute_result,
# Why the loop stopped. triggered=False alone cannot tell a refusal from a run
# that hit max_iterations while still on track -- see LoopExit.
exit_reason=best.exit_reason.value,
)


def evaluate_agentic_anomaly_detection(
Expand Down Expand Up @@ -630,22 +648,24 @@ def _write_scores(ctx: RunTraceContext) -> None:
f"Analysed {detail['actual_metrics']} at {detail['actual_granularity']}. "
f"Actual execute result: {best.actual_execute_result}."
)
exc = AnomalyDetectionAssertionError(message)
exc.reasoning_steps = best.reasoning_steps
exc.conversation_id = best.conversation_id
exc.response_id = best.response_id
exc.detail = detail
exc.runs_passed = runs_passed
exc.runs_effective = len(summary.run_results)
exc.best_run_latency_s = best.run_latency_s
raise exc

return AgenticEvalOutcome(
runs_passed=runs_passed,
runs_effective=len(summary.run_results),
raise_agentic_failure(
AnomalyDetectionAssertionError,
message,
reasoning_steps=best.reasoning_steps,
conversation_id=best.conversation_id,
response_id=best.response_id,
detail=detail,
runs_passed=runs_passed,
runs_effective=len(summary.run_results),
best_run_latency_s=best.run_latency_s,
)

return agentic_success(
reasoning_steps=best.reasoning_steps,
conversation_id=best.conversation_id,
response_id=best.response_id,
detail=detail,
runs_passed=runs_passed,
runs_effective=len(summary.run_results),
best_run_latency_s=best.run_latency_s,
)
Loading