diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py new file mode 100644 index 000000000..9c35413e8 --- /dev/null +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py @@ -0,0 +1,109 @@ +# (C) 2026 GoodData Corporation +"""The common tail of every ``evaluate_agentic_*`` function. + +Before this module, each of the agentic evaluators hand-wrote the same block: copy +``reasoning_steps``/``conversation_id``/``response_id``/``detail``/``runs_passed``/ +``runs_effective``/``best_run_latency_s`` (and sometimes ``timings``) either onto a +raised exception or into the returned ``AgenticEvalOutcome``. Eleven independent copies +of the same ~7 lines meant a new universal field needed eleven edits, not one, and +nothing failed loudly when a copy was missed -- exactly what happened to ``timings`` +(see ``AgenticAssertionError``'s docstring: "while it lived in eight copies, two +declared timings and six did not") and again to ``best_run_latency_s`` (every agentic +kind silently reported it as ``null`` until this was noticed and fixed kind-by-kind). + +``agentic_detail`` has the same motivation for the ``detail`` dict's timeline fields: +``timeline_detail`` builds both ``latency_breakdown`` and ``tool_calls`` from the same +events so they stay index-aligned, but roughly half the evaluators called +``build_latency_breakdown`` directly and silently never got a ``tool_calls`` key. +""" + +from __future__ import annotations + +from typing import Any, NoReturn + +from gooddata_eval.core.models import ( + AgenticAssertionError, + AgenticEvalOutcome, + ReasoningStepEvent, + ToolCallEvent, + timeline_detail, +) +from gooddata_eval.core.timing import PhaseTimings + +__all__ = ["agentic_detail", "agentic_success", "raise_agentic_failure"] + + +def agentic_detail( + tool_call_events: list[ToolCallEvent], + reasoning_step_events: list[ReasoningStepEvent] | None, + **kind_specific: Any, +) -> dict: + """A kind's full ``detail`` dict: its own fields plus the universal timeline ones. + + ``kind_specific`` comes first in the merge so a kind can never accidentally shadow + ``latency_breakdown``/``tool_calls`` with a same-named field of its own. + """ + return {**kind_specific, **timeline_detail(tool_call_events, reasoning_step_events)} + + +def raise_agentic_failure( + exception_cls: type[AgenticAssertionError], + message: str, + *, + reasoning_steps: list[str], + conversation_id: str, + response_id: str | None, + detail: dict, + runs_passed: int, + runs_effective: int, + best_run_latency_s: float | None, + timings: PhaseTimings | None = None, +) -> NoReturn: + """Build ``exception_cls(message)`` with every common field attached, and raise it. + + ``exception_cls`` must be an ``AgenticAssertionError`` subclass -- that base class is + what declares these fields as legal targets (see its docstring). A kind whose own + "no verdict at all" branch raises a bare ``JudgeResponseError`` instead (not a subclass) + does not go through this helper for that branch. + + ``timings`` stays optional: only the three kinds that track per-run ``PhaseTimings`` + (general_question, metric_skill, dashboard_skill) pass one -- the rest keep their + own ``PhaseTimings()`` zero default, same as before this helper existed. + """ + exc = exception_cls(message) + exc.reasoning_steps = reasoning_steps + exc.conversation_id = conversation_id + exc.response_id = response_id + exc.detail = detail + exc.runs_passed = runs_passed + exc.runs_effective = runs_effective + exc.best_run_latency_s = best_run_latency_s + if timings is not None: + exc.timings = timings + raise exc + + +def agentic_success( + *, + reasoning_steps: list[str], + conversation_id: str | None, + response_id: str | None, + detail: dict, + runs_passed: int, + runs_effective: int, + best_run_latency_s: float | None, + timings: PhaseTimings | None = None, +) -> AgenticEvalOutcome: + """The success-path mirror of ``raise_agentic_failure`` -- same fields, same shape.""" + kwargs: dict[str, Any] = { + "reasoning_steps": reasoning_steps, + "conversation_id": conversation_id, + "response_id": response_id, + "detail": detail, + "runs_passed": runs_passed, + "runs_effective": runs_effective, + "best_run_latency_s": best_run_latency_s, + } + if timings is not None: + kwargs["timings"] = timings + return AgenticEvalOutcome(**kwargs) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py index bb59fdf45..45c9d1bc1 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py @@ -21,6 +21,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -40,7 +41,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) try: @@ -951,28 +951,30 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.eval - detail = { - "alert_created": ev.alert_created, - "operator_correct": ev.operator_correct, - "threshold_correct": ev.threshold_correct, - "trigger_correct": ev.trigger_correct, - "filters_correct": ev.filters_correct, - "metric_correct": ev.metric_correct, - "recipients_correct": ev.recipients_correct, - "attributes_correct": ev.attributes_correct, - "granularity_correct": ev.granularity_correct, - "actual_alert_arguments": best.actual_alert_arguments, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + alert_created=ev.alert_created, + operator_correct=ev.operator_correct, + threshold_correct=ev.threshold_correct, + trigger_correct=ev.trigger_correct, + filters_correct=ev.filters_correct, + metric_correct=ev.metric_correct, + recipients_correct=ev.recipients_correct, + attributes_correct=ev.attributes_correct, + granularity_correct=ev.granularity_correct, + actual_alert_arguments=best.actual_alert_arguments, # Why the loop stopped. alert_created=False alone cannot tell a refusal from a run # that hit max_iterations while still on track -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) - exc = AlertSkillAssertionError( + raise_agentic_failure( + AlertSkillAssertionError, f"Alert skill assertion failed. {gate_note} strict_pass={ev.strict_pass}. " f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, " f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, " @@ -980,22 +982,21 @@ def _write_scores(ctx: RunTraceContext) -> None: f"recipients_correct={ev.recipients_correct}, " f"attributes_correct={ev.attributes_correct}, " f"granularity_correct={ev.granularity_correct}. " - f"Actual args: {best.actual_alert_arguments}" + f"Actual args: {best.actual_alert_arguments}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py index daa603064..a2b4050ee 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py @@ -40,6 +40,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -56,9 +57,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -287,6 +288,10 @@ class AnomalyRunResult: # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from # turn_wall_clock_sec above, which is only the final triggering turn. run_latency_s: float = 0.0 + # Why the simulated-user loop stopped -- see LoopExit. `triggered=False` alone cannot + # separate a refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -391,6 +396,10 @@ def _accumulate(result: ChatResult) -> None: all_tool_call_events.extend(result.tool_call_events or []) all_reasoning_step_events.extend(result.reasoning_step_events or []) + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED + for iteration in range(max_iterations): try: chat_result = client.send_message(conv_id, current_question) @@ -403,6 +412,7 @@ def _accumulate(result: ChatResult) -> None: _accumulate(partial) viz_args, execute_result = _extract_anomaly_calls(all_tool_call_events) turn_completed = False + exit_reason = LoopExit.CHAT_ERROR break reasoning_steps.extend(chat_result.reasoning_steps or []) response_id = chat_result.response_id or response_id @@ -416,8 +426,10 @@ def _accumulate(result: ChatResult) -> None: if execute_result is not None: # The turn that ran the detection, not an earlier disambiguation turn. turn_wall_clock_sec = chat_result.turn_wall_clock_sec + exit_reason = LoopExit.SUCCESS break if not response_text: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -426,6 +438,7 @@ def _accumulate(result: ChatResult) -> None: disambiguated = True except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run _log.warning("Simulated anomaly user reply failed for conversation %s: %s", conv_id, exc) + exit_reason = LoopExit.SIMULATED_USER_FAILED break return AnomalyRunResult( @@ -434,6 +447,7 @@ def _accumulate(result: ChatResult) -> None: actual_visualization=viz_args, actual_execute_result=execute_result, turn_wall_clock_sec=turn_wall_clock_sec, + exit_reason=exit_reason, reasoning_steps=reasoning_steps, response_id=response_id, tool_call_events=all_tool_call_events, @@ -487,25 +501,29 @@ class AnomalyDetectionAssertionError(AgenticAssertionError): def _detail(best: AnomalyRunResult) -> dict[str, Any]: ev = best.evaluation - return { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "metric_correct": ev.metric_correct, - "granularity_correct": ev.granularity_correct, + return agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + metric_correct=ev.metric_correct, + granularity_correct=ev.granularity_correct, # Which content checks the fixture pinned -- without it a run that verified nothing # reads the same as one where everything matched. - "asserted": ev.asserted, - "disambiguated": ev.disambiguated, - "actual_metrics": sorted(_metric_uris(best.actual_visualization)), - "actual_granularity": _inferred_granularity(best.actual_visualization), + asserted=ev.asserted, + disambiguated=ev.disambiguated, + actual_metrics=sorted(_metric_uris(best.actual_visualization)), + actual_granularity=_inferred_granularity(best.actual_visualization), # Reported, never asserted: whether a real series contains anomalies is a property # of the data, so a fixture demanding some would fail on the next warehouse refresh. - "anomaly_point_count": _point_count(best.actual_execute_result), - "actual_execute_result": best.actual_execute_result, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + anomaly_point_count=_point_count(best.actual_execute_result), + actual_execute_result=best.actual_execute_result, + # Why the loop stopped. triggered=False alone cannot tell a refusal from a run + # that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) def evaluate_agentic_anomaly_detection( @@ -630,22 +648,24 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Analysed {detail['actual_metrics']} at {detail['actual_granularity']}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = AnomalyDetectionAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = len(summary.run_results) - exc.best_run_latency_s = best.run_latency_s - raise exc - - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=len(summary.run_results), + raise_agentic_failure( + AnomalyDetectionAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, + ) + + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py index 349adcf77..e41cc271f 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py @@ -41,6 +41,7 @@ summarize_visualizations, ) from gooddata_eval.core.agentic._gate import log_gate_scores +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -66,7 +67,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.scoring import ( check_filters, @@ -1023,21 +1023,22 @@ def run_agentic_conversation( def _conversation_detail(result: ConversationResult) -> dict: - return { - "mode": result.mode, - "full_skill_coverage": result.full_skill_coverage, - "conversation_success": result.conversation_success, - "context_success": result.context_success, - "context_kept_rate": result.context_kept_rate, - "turns_before_first_break": result.turns_before_first_break, - "lost_context_clarifications": result.lost_context_clarifications, - "stalled_turns": result.stalled_turns, - "total_clarification_turns": result.total_clarification_turns, - "max_clarification_turns": result.max_clarification_turns, - "judge_model": result.judge_model, - "turns": [tr.detail() for tr in result.turn_results], - **timeline_detail(result.tool_call_events, result.reasoning_step_events), - } + return agentic_detail( + result.tool_call_events, + result.reasoning_step_events, + mode=result.mode, + full_skill_coverage=result.full_skill_coverage, + conversation_success=result.conversation_success, + context_success=result.context_success, + context_kept_rate=result.context_kept_rate, + turns_before_first_break=result.turns_before_first_break, + lost_context_clarifications=result.lost_context_clarifications, + stalled_turns=result.stalled_turns, + total_clarification_turns=result.total_clarification_turns, + max_clarification_turns=result.max_clarification_turns, + judge_model=result.judge_model, + turns=[tr.detail() for tr in result.turn_results], + ) class ConversationAssertionError(AgenticAssertionError): @@ -1215,18 +1216,20 @@ def _write_scores(ctx: RunTraceContext) -> None: f"full_skill_coverage={result.full_skill_coverage}. " f"Failed turns: {legacy_failed}" ) - exc = ConversationAssertionError(message) - exc.reasoning_steps = result.reasoning_steps - exc.conversation_id = result.conversation_id - exc.response_id = result.response_id - exc.detail = detail # This kind takes no k and drives its fixture exactly once, whatever --runs asks # for. Saying so explicitly stops the report claiming K runs that never happened. - exc.runs_passed = 0 - exc.runs_effective = 1 - exc.best_run_latency_s = result.run_latency_s - raise exc - return AgenticEvalOutcome( + raise_agentic_failure( + ConversationAssertionError, + message, + reasoning_steps=result.reasoning_steps, + conversation_id=result.conversation_id, + response_id=result.response_id, + detail=detail, + runs_passed=0, + runs_effective=1, + best_run_latency_s=result.run_latency_s, + ) + return agentic_success( reasoning_steps=result.reasoning_steps, conversation_id=result.conversation_id, response_id=result.response_id, diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py index f33cc9dae..29cc1892d 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py @@ -17,6 +17,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -33,9 +34,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings @@ -664,6 +665,11 @@ class DashboardRunResult: tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) timings: PhaseTimings = field(default_factory=PhaseTimings) + # Why the loop stopped -- see LoopExit. A dashboard not produced alone cannot tell a + # refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). No CHAT_ERROR/SIMULATED_USER_FAILED member here: this + # loop has neither a send_message try/except nor a simulated-user call to fail. + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -864,6 +870,11 @@ def _execute_single_dashboard_run( turn_offset = 0.0 tool_index_offset = 0 reasoning_index_offset = 0 + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() (or the is_edit "no reply of its own" branch below, which + # is the same shape of outcome -- the agent answered, just not with the dashboard tool) + # is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED for iteration in range(max_iterations): turns += 1 @@ -903,10 +914,12 @@ def _execute_single_dashboard_run( dashboard_part = _extract_dashboard_part(chat_result, "dashboard") if is_edit: patch_part = _extract_dashboard_part(chat_result, _PATCH_TYPE) + exit_reason = LoopExit.SUCCESS break response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) if not response_text and not chat_result.tool_call_events: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -941,6 +954,7 @@ def _execute_single_dashboard_run( tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, timings=timings, + exit_reason=exit_reason, ) @@ -1100,13 +1114,17 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail: dict[str, Any] = { + detail: dict[str, Any] = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, **best.evaluation.strict_checks, **best.evaluation.diagnostics, - "failures": best.evaluation.failures, - "notes": best.evaluation.notes, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + failures=best.evaluation.failures, + notes=best.evaluation.notes, + # Why the loop stopped. A dashboard not produced alone cannot tell a refusal from + # a run that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -1127,28 +1145,28 @@ def _write_scores(ctx: RunTraceContext) -> None: " one feature flag registers both, so check that before the model." ) notes = "; ".join(best.evaluation.notes) - exc = DashboardSkillAssertionError( + raise_agentic_failure( + DashboardSkillAssertionError, f"Dashboard skill assertion failed. {gate_note}{skill_note} " f"Checks: {best.evaluation.strict_checks}. " f"Failures: {'; '.join(best.evaluation.failures) or 'none reported'}." - + (f" Notes (not scored): {notes}." if notes else "") + + (f" Notes (not scored): {notes}." if notes else ""), + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.best_run_latency_s = run_latency_s(best.timings) - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, - timings=item_timings, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py index 6db82dbda..b54e772e5 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py @@ -14,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -32,7 +33,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, ) from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings @@ -325,38 +325,39 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail = { - "judge_passed": best.passed, - "judge_reasoning": best.reasoning, - "actual_output": best.actual_output, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + judge_passed=best.passed, + judge_reasoning=best.reasoning, + actual_output=best.actual_output, # Only present when it happened, so the usual JSON shape is unchanged. A # pass@K computed over fewer runs than --runs asked for is a weaker result and # the report has to say so. **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), - } + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - exc = GeneralQuestionAssertionError( - f"General question assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}" + raise_agentic_failure( + GeneralQuestionAssertionError, + f"General question assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.best_run_latency_s = run_latency_s(best.timings) - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, - timings=item_timings, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py index bc5a4f4d0..4c1d9d2fd 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py @@ -14,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -33,7 +34,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - timeline_detail, ) _DEFAULT_K = 1 @@ -301,35 +301,36 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail = { - "judge_passed": best.passed, - "judge_reasoning": best.reasoning, - "actual_output": best.actual_output, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + judge_passed=best.passed, + judge_reasoning=best.reasoning, + actual_output=best.actual_output, # Only present when it happened, so the usual JSON shape is unchanged. A # pass@K over fewer runs than --runs asked for is a weaker result. **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), - } + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - exc = GuardrailAssertionError( - f"Guardrail assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}" + raise_agentic_failure( + GuardrailAssertionError, + f"Guardrail assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py index 7b29bf624..01d4ae04f 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py @@ -18,6 +18,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -37,7 +38,6 @@ LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -528,20 +528,21 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.evaluation - detail = { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "disambiguated": ev.disambiguated, - "actual_create_args": best.actual_create_args, - "actual_execute_result": best.actual_execute_result, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + disambiguated=ev.disambiguated, + actual_create_args=best.actual_create_args, + actual_execute_result=best.actual_execute_result, # Why the loop stopped -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -552,21 +553,23 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Actual create args: {best.actual_create_args}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = KdaSkillAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + raise_agentic_failure( + KdaSkillAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, + ) + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py index e2f95109e..a4d3c9e88 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py @@ -19,6 +19,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -39,7 +40,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings @@ -549,44 +549,45 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best expected_outputs_list: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output] - detail = { - "metric_created": best.metric_created, - "maql_correct": best.maql_correct, - "expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list], - "actual_maql": best.actual_maql, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + metric_created=best.metric_created, + maql_correct=best.maql_correct, + expected_maql_candidates=[c.get("maql", "") for c in expected_outputs_list], + actual_maql=best.actual_maql, # Why the loop stopped. metric_created=False alone cannot tell a refusal from a run # that hit max_iterations while still on track -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) candidates_str = "; ".join(repr(c.get("maql", "")) for c in expected_outputs_list) - exc = MetricSkillAssertionError( + raise_agentic_failure( + MetricSkillAssertionError, f"Metric skill assertion failed. {gate_note} " f"metric_created={best.metric_created}, maql_correct={best.maql_correct}. " f"Expected MAQL (candidates): {candidates_str}. " - f"Actual MAQL: {best.actual_maql}." + f"Actual MAQL: {best.actual_maql}.", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.best_run_latency_s = run_latency_s(best.timings) - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, - timings=item_timings, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py index 95fd9ef8a..a11c3cf24 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py @@ -14,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -30,7 +31,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, ) _DEFAULT_K = 1 @@ -263,34 +263,35 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail = { - "tool_selected": best.tool_selected, - "tool_correct": best.tool_correct, - "tool_call_names": best.tool_call_names, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + tool_selected=best.tool_selected, + tool_correct=best.tool_correct, + tool_call_names=best.tool_call_names, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) - exc = SearchToolAssertionError( + raise_agentic_failure( + SearchToolAssertionError, f"Search tool assertion failed. {gate_note} " f"tool_selected={best.tool_selected}, tool_correct={best.tool_correct}. " - f"Tool calls made: {best.tool_call_names}" + f"Tool calls made: {best.tool_call_names}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py index 9ad64b722..382b12877 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py @@ -19,6 +19,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -46,7 +47,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.scoring import get_dimension_uri_set, get_metric_uri_set, uri_to_display_name @@ -485,15 +485,16 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.eval_result - detail = { + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, **evaluation_result_detail(ev), # Why the loop stopped -- see LoopExit. total_turns is already the turn count for # this run, so it doubles as turns_used. - "exit_reason": best.exit_reason.value, - "turns_used": int(best.total_turns), - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=int(best.total_turns), + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -502,7 +503,8 @@ def _write_scores(ctx: RunTraceContext) -> None: cross_ref_detail = (" → " + "; ".join(ev.cross_ref_errors)) if ev.cross_ref_errors else "" expected_dump = best.best_expected.model_dump(exclude_none=True) actual_dump = best.actual_output.model_dump(exclude_none=True) if best.actual_output else None - exc = VisualizationAssertionError( + raise_agentic_failure( + VisualizationAssertionError, "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" "Agentic Visualization Assertion Failed! (Critical Mode)\n" f"{gate_note}\n" @@ -530,22 +532,21 @@ def _write_scores(ctx: RunTraceContext) -> None: f" attribute : {ev.filter_attribute_score}\n" f"{_filter_diff('attribute', ev)}" f" Viz Type Hard : {ev.viz_type_hard}\n" - "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" + "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.best_run_latency_s = best.run_latency_s - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py index d51ac6fde..255f369dc 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py @@ -26,6 +26,7 @@ from dataclasses import dataclass, field from typing import Any +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -43,9 +44,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -206,6 +207,10 @@ class WhatIfRunResult: # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from # turn_wall_clock_sec above, which is only the scenario-execution turn. run_latency_s: float = 0.0 + # Why the simulated-user loop stopped -- see LoopExit. `triggered=False` alone cannot + # separate a refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -340,6 +345,10 @@ def _accumulate(result: ChatResult) -> None: all_tool_call_events.extend(result.tool_call_events or []) all_reasoning_step_events.extend(result.reasoning_step_events or []) + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED + for iteration in range(max_iterations): try: chat_result = client.send_message(conv_id, current_question) @@ -352,6 +361,7 @@ def _accumulate(result: ChatResult) -> None: _accumulate(partial) create_args, execute_result = _extract_what_if_calls(all_tool_call_events) turn_completed = False + exit_reason = LoopExit.CHAT_ERROR break reasoning_steps.extend(chat_result.reasoning_steps or []) response_id = chat_result.response_id or response_id @@ -365,8 +375,10 @@ def _accumulate(result: ChatResult) -> None: if execute_result is not None: # The turn that ran the scenario, not an earlier disambiguation turn. turn_wall_clock_sec = chat_result.turn_wall_clock_sec + exit_reason = LoopExit.SUCCESS break if not response_text: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -375,6 +387,7 @@ def _accumulate(result: ChatResult) -> None: disambiguated = True except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run _log.warning("Simulated what-if user reply failed for conversation %s: %s", conv_id, exc) + exit_reason = LoopExit.SIMULATED_USER_FAILED break return WhatIfRunResult( @@ -382,6 +395,7 @@ def _accumulate(result: ChatResult) -> None: evaluation=_evaluate_run(create_args, execute_result, expected_output, turn_completed, disambiguated), actual_create_args=create_args, actual_execute_result=execute_result, + exit_reason=exit_reason, turn_wall_clock_sec=turn_wall_clock_sec, reasoning_steps=reasoning_steps, response_id=response_id, @@ -439,26 +453,30 @@ class WhatIfAssertionError(AgenticAssertionError): def _detail(best: WhatIfRunResult) -> dict[str, Any]: ev = best.evaluation adjustments = _adjustments(best.actual_create_args) - return { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "metric_correct": ev.metric_correct, - "maql_correct": ev.maql_correct, - "scenario_count_correct": ev.scenario_count_correct, - "baseline_correct": ev.baseline_correct, + return agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + metric_correct=ev.metric_correct, + maql_correct=ev.maql_correct, + scenario_count_correct=ev.scenario_count_correct, + baseline_correct=ev.baseline_correct, # Which content checks the fixture pinned -- without it a run that verified nothing # reads the same as one where everything matched. - "asserted": ev.asserted, - "disambiguated": ev.disambiguated, - "actual_adjustments": adjustments, - "actual_scenario_labels": [ + asserted=ev.asserted, + disambiguated=ev.disambiguated, + actual_adjustments=adjustments, + actual_scenario_labels=[ s.get("label") for s in ((best.actual_create_args or {}).get("scenarios") or []) if isinstance(s, dict) ], - "actual_execute_result": best.actual_execute_result, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + actual_execute_result=best.actual_execute_result, + # Why the loop stopped. triggered=False alone cannot tell a refusal from a run + # that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) def evaluate_agentic_what_if( @@ -581,22 +599,24 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Actual adjustments: {detail['actual_adjustments']}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = WhatIfAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = len(summary.run_results) - exc.best_run_latency_s = best.run_latency_s - raise exc - - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=len(summary.run_results), + raise_agentic_failure( + WhatIfAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, + ) + + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py b/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py index 84e138fb1..f3aa69534 100644 --- a/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py +++ b/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py @@ -16,7 +16,7 @@ evaluate_agentic_anomaly_detection, run_agentic_anomaly_detection, ) -from gooddata_eval.core.models import ChatResult +from gooddata_eval.core.models import ChatResult, LoopExit _MODULE = "gooddata_eval.core.agentic.anomaly_detection" @@ -254,18 +254,22 @@ def test_a_single_turn_detection_passes(): summary = _run([_chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is False + assert summary.best.exit_reason is LoopExit.SUCCESS def test_a_clarifying_question_is_answered_and_the_run_continues(): summary = _run([_chat([], text="Which Spend metric did you mean?"), _chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is True + assert summary.best.exit_reason is LoopExit.SUCCESS def test_the_loop_stops_at_max_iterations_without_a_detection(): + # executed=False alone cannot tell this apart from a refusal -- exit_reason can. summary = _run([_chat([], text="Still thinking")] * 4, max_iterations=4) assert summary.pass_at_k is False assert summary.best.evaluation.executed is False + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_response_ends_the_run_immediately(): @@ -276,17 +280,19 @@ def test_an_empty_response_ends_the_run_immediately(): patch(f"{_MODULE}.ChatClient", return_value=client), patch(f"{_MODULE}.generate_simulated_anomaly_response") as sim, ): - run_agentic_anomaly_detection( + summary = run_agentic_anomaly_detection( host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED ) assert client.send_message.call_count == 1 sim.assert_not_called() + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_chat_error_on_a_later_run_does_not_discard_the_earlier_one(): summary = _run([_chat(_pair()), RuntimeError("boom")], k=2) assert len(summary.run_results) == 2 assert summary.pass_at_k is True + assert summary.run_results[1].exit_reason is LoopExit.CHAT_ERROR def test_k_must_be_at_least_one(): @@ -318,6 +324,8 @@ def test_detail_reports_the_series_analysed_and_the_flag_count(): assert outcome.detail["anomaly_point_count"] == 0 assert outcome.detail["asserted"] == ["metric", "granularity"] assert "latency_breakdown" in outcome.detail + assert "tool_calls" in outcome.detail + assert outcome.detail["exit_reason"] == "success" assert outcome.runs_passed == 1 diff --git a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py index e949d8a3b..49620a8e8 100644 --- a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py @@ -15,7 +15,7 @@ evaluate_dashboard_response, run_agentic_dashboard_skill, ) -from gooddata_eval.core.models import ChatResult, ToolCallEvent +from gooddata_eval.core.models import ChatResult, LoopExit, ToolCallEvent # Ids and titles are the ones the eval layout seeds; shapes are trimmed from real runs of # `agent_dashboard_skill` against ecommerce_demo. @@ -1086,6 +1086,8 @@ def test_the_loop_is_capped(self): assert client.send_message.call_count == 2 assert not summary.pass_at_k assert not summary.best.evaluation.drafted + # drafted=False alone cannot tell this apart from a refusal -- exit_reason can. + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_turn_stops_the_loop(self): client = MagicMock() @@ -1093,6 +1095,7 @@ def test_an_empty_turn_stops_the_loop(self): summary = _run_with(client, _DC05_EXPECTED, max_iterations=5) assert client.send_message.call_count == 1 assert not summary.pass_at_k + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_caller_supplied_conversation_is_not_deleted(self): client = MagicMock() @@ -1161,6 +1164,8 @@ def test_a_passing_run_returns_the_outcome(self): assert outcome.runs_passed == 1 assert outcome.runs_effective == 1 assert outcome.detail["charts_matched"] is True + assert outcome.detail["exit_reason"] == "success" + assert "tool_calls" in outcome.detail def _as_events(raw: list[dict]) -> list[ToolCallEvent]: diff --git a/packages/gooddata-eval/tests/test_agentic_general_question.py b/packages/gooddata-eval/tests/test_agentic_general_question.py index cd9db187e..26dddbacc 100644 --- a/packages/gooddata-eval/tests/test_agentic_general_question.py +++ b/packages/gooddata-eval/tests/test_agentic_general_question.py @@ -257,6 +257,7 @@ def test_evaluate_agentic_general_question_returns_reasoning_steps_on_pass(): "judge_reasoning": "Correct answer", "actual_output": "42", "latency_breakdown": [], + "tool_calls": [], } @@ -284,6 +285,7 @@ def test_evaluate_agentic_general_question_attaches_reasoning_steps_to_exception "judge_reasoning": "Wrong answer", "actual_output": "I don't know", "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_kda_skill.py b/packages/gooddata-eval/tests/test_agentic_kda_skill.py index 573a2a267..3d6f5b620 100644 --- a/packages/gooddata-eval/tests/test_agentic_kda_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_kda_skill.py @@ -1134,6 +1134,7 @@ def test_evaluate_agentic_kda_skill_returns_reasoning_steps_on_pass(): "turns_used": 1, "max_iterations": 1, "latency_breakdown": [], + "tool_calls": [], } @@ -1172,6 +1173,7 @@ def test_evaluate_agentic_kda_skill_attaches_reasoning_steps_to_exception_on_fai "turns_used": 1, "max_iterations": 1, "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_search_tool.py b/packages/gooddata-eval/tests/test_agentic_search_tool.py index 975004cd4..f1100d9c9 100644 --- a/packages/gooddata-eval/tests/test_agentic_search_tool.py +++ b/packages/gooddata-eval/tests/test_agentic_search_tool.py @@ -208,6 +208,7 @@ def test_evaluate_agentic_search_tool_returns_reasoning_steps_on_pass(): "tool_correct": True, "tool_call_names": ["search_objects"], "latency_breakdown": [], + "tool_calls": [], } @@ -244,4 +245,5 @@ def test_evaluate_agentic_search_tool_attaches_reasoning_steps_to_exception_on_f "tool_correct": False, "tool_call_names": [], "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_what_if.py b/packages/gooddata-eval/tests/test_agentic_what_if.py index 07da79ca8..baf267442 100644 --- a/packages/gooddata-eval/tests/test_agentic_what_if.py +++ b/packages/gooddata-eval/tests/test_agentic_what_if.py @@ -13,7 +13,7 @@ evaluate_agentic_what_if, run_agentic_what_if, ) -from gooddata_eval.core.models import ChatResult +from gooddata_eval.core.models import ChatResult, LoopExit _MODULE = "gooddata_eval.core.agentic.what_if" @@ -215,6 +215,7 @@ def test_a_single_turn_scenario_passes(): summary = _run([_chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is False + assert summary.best.exit_reason is LoopExit.SUCCESS def test_a_clarifying_question_is_answered_and_the_run_continues(): @@ -222,12 +223,15 @@ def test_a_clarifying_question_is_answered_and_the_run_continues(): summary = _run([_chat([], text="Which Spend metric did you mean?"), _chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is True + assert summary.best.exit_reason is LoopExit.SUCCESS def test_the_loop_stops_at_max_iterations_without_a_scenario(): + # triggered=False alone cannot tell this apart from a refusal -- exit_reason can. summary = _run([_chat([], text="Still thinking")] * 4, max_iterations=4) assert summary.pass_at_k is False assert summary.best.evaluation.triggered is False + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_response_ends_the_run_immediately(): @@ -238,15 +242,19 @@ def test_an_empty_response_ends_the_run_immediately(): patch(f"{_MODULE}.ChatClient", return_value=client), patch(f"{_MODULE}.generate_simulated_what_if_response") as sim, ): - run_agentic_what_if(host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED) + summary = run_agentic_what_if( + host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED + ) assert client.send_message.call_count == 1 sim.assert_not_called() + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_chat_error_on_a_later_run_does_not_discard_the_earlier_one(): summary = _run([_chat(_pair()), RuntimeError("boom")], k=2) assert len(summary.run_results) == 2 assert summary.pass_at_k is True + assert summary.run_results[1].exit_reason is LoopExit.CHAT_ERROR def test_k_must_be_at_least_one(): @@ -277,6 +285,8 @@ def test_detail_reports_the_adjustments_the_agent_asked_for(): assert outcome.detail["actual_scenario_labels"] == ["Scenario A"] assert outcome.detail["asserted"] == ["metric_id", "scenario_maql"] assert "latency_breakdown" in outcome.detail + assert "tool_calls" in outcome.detail + assert outcome.detail["exit_reason"] == "success" assert outcome.runs_passed == 1