diff --git a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py index cdef740b7..5f3a89e86 100644 --- a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py @@ -369,6 +369,18 @@ def _apply_timings(item_report: ItemReport, timings: Any) -> None: item_report.simulated_user_latency_s = timings.simulated_user_s +def _apply_best_run_latency(item_report: ItemReport, source: Any) -> None: + """Copy the rank-selected run's own wall time, mirroring the single-shot path's + ``ItemReport.best_run_latency_s`` (see ``core/runner.py``'s ``_run_one_item``). + + Kinds not yet wired to measure it pass None and keep the field at its own None + default rather than reporting an invented number. + """ + best_run_latency_s = getattr(source, "best_run_latency_s", None) + if best_run_latency_s is not None: + item_report.best_run_latency_s = best_run_latency_s + + def run_agentic_items( items: list[DatasetItem], host: str, @@ -454,6 +466,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport: item_report.response_id = response_id item_report.best_detail = detail or {} _apply_timings(item_report, getattr(outcome, "timings", None)) + _apply_best_run_latency(item_report, outcome) _apply_run_counts(item_report, outcome) except AssertionError as exc: item_report.gate_passed = False if gated else None @@ -463,6 +476,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport: item_report.response_id = getattr(exc, "response_id", None) item_report.best_detail = getattr(exc, "detail", None) or {} _apply_timings(item_report, getattr(exc, "timings", None)) + _apply_best_run_latency(item_report, exc) _apply_run_counts(item_report, exc) # Read off the counts, not off the gate: pass^K fails items where runs did pass, # and reporting those as pass_at_k False would contradict the Langfuse score of @@ -477,6 +491,7 @@ def _process_item(index: int, item: DatasetItem) -> ItemReport: # judge broke should not also report the agent as costing 0s. Kinds that # attach no timings to the exception keep their 0.0 defaults. _apply_timings(item_report, getattr(exc, "timings", None)) + _apply_best_run_latency(item_report, exc) finally: item_report.latency_s = time.perf_counter() - t0 diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py new file mode 100644 index 000000000..9c35413e8 --- /dev/null +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/_outcome.py @@ -0,0 +1,109 @@ +# (C) 2026 GoodData Corporation +"""The common tail of every ``evaluate_agentic_*`` function. + +Before this module, each of the agentic evaluators hand-wrote the same block: copy +``reasoning_steps``/``conversation_id``/``response_id``/``detail``/``runs_passed``/ +``runs_effective``/``best_run_latency_s`` (and sometimes ``timings``) either onto a +raised exception or into the returned ``AgenticEvalOutcome``. Eleven independent copies +of the same ~7 lines meant a new universal field needed eleven edits, not one, and +nothing failed loudly when a copy was missed -- exactly what happened to ``timings`` +(see ``AgenticAssertionError``'s docstring: "while it lived in eight copies, two +declared timings and six did not") and again to ``best_run_latency_s`` (every agentic +kind silently reported it as ``null`` until this was noticed and fixed kind-by-kind). + +``agentic_detail`` has the same motivation for the ``detail`` dict's timeline fields: +``timeline_detail`` builds both ``latency_breakdown`` and ``tool_calls`` from the same +events so they stay index-aligned, but roughly half the evaluators called +``build_latency_breakdown`` directly and silently never got a ``tool_calls`` key. +""" + +from __future__ import annotations + +from typing import Any, NoReturn + +from gooddata_eval.core.models import ( + AgenticAssertionError, + AgenticEvalOutcome, + ReasoningStepEvent, + ToolCallEvent, + timeline_detail, +) +from gooddata_eval.core.timing import PhaseTimings + +__all__ = ["agentic_detail", "agentic_success", "raise_agentic_failure"] + + +def agentic_detail( + tool_call_events: list[ToolCallEvent], + reasoning_step_events: list[ReasoningStepEvent] | None, + **kind_specific: Any, +) -> dict: + """A kind's full ``detail`` dict: its own fields plus the universal timeline ones. + + ``kind_specific`` comes first in the merge so a kind can never accidentally shadow + ``latency_breakdown``/``tool_calls`` with a same-named field of its own. + """ + return {**kind_specific, **timeline_detail(tool_call_events, reasoning_step_events)} + + +def raise_agentic_failure( + exception_cls: type[AgenticAssertionError], + message: str, + *, + reasoning_steps: list[str], + conversation_id: str, + response_id: str | None, + detail: dict, + runs_passed: int, + runs_effective: int, + best_run_latency_s: float | None, + timings: PhaseTimings | None = None, +) -> NoReturn: + """Build ``exception_cls(message)`` with every common field attached, and raise it. + + ``exception_cls`` must be an ``AgenticAssertionError`` subclass -- that base class is + what declares these fields as legal targets (see its docstring). A kind whose own + "no verdict at all" branch raises a bare ``JudgeResponseError`` instead (not a subclass) + does not go through this helper for that branch. + + ``timings`` stays optional: only the three kinds that track per-run ``PhaseTimings`` + (general_question, metric_skill, dashboard_skill) pass one -- the rest keep their + own ``PhaseTimings()`` zero default, same as before this helper existed. + """ + exc = exception_cls(message) + exc.reasoning_steps = reasoning_steps + exc.conversation_id = conversation_id + exc.response_id = response_id + exc.detail = detail + exc.runs_passed = runs_passed + exc.runs_effective = runs_effective + exc.best_run_latency_s = best_run_latency_s + if timings is not None: + exc.timings = timings + raise exc + + +def agentic_success( + *, + reasoning_steps: list[str], + conversation_id: str | None, + response_id: str | None, + detail: dict, + runs_passed: int, + runs_effective: int, + best_run_latency_s: float | None, + timings: PhaseTimings | None = None, +) -> AgenticEvalOutcome: + """The success-path mirror of ``raise_agentic_failure`` -- same fields, same shape.""" + kwargs: dict[str, Any] = { + "reasoning_steps": reasoning_steps, + "conversation_id": conversation_id, + "response_id": response_id, + "detail": detail, + "runs_passed": runs_passed, + "runs_effective": runs_effective, + "best_run_latency_s": best_run_latency_s, + } + if timings is not None: + kwargs["timings"] = timings + return AgenticEvalOutcome(**kwargs) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py index af9a7090b..bdfa7b822 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py @@ -6,6 +6,7 @@ import json import os import re +import time from dataclasses import dataclass, field from typing import Any @@ -20,6 +21,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -39,7 +41,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) try: @@ -490,6 +491,9 @@ class AlertRunResult: # also report operator/threshold/metric/recipients as False. exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED turns_used: int = 0 + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -676,6 +680,7 @@ def run_agentic_alert_skill( def _run_once(conv_id: str) -> AlertRunResult: alert_id_to_delete: str | None = None + run_started = time.monotonic() try: alert_id: str | None = None actual_args: dict = {} @@ -794,6 +799,7 @@ def _run_once(conv_id: str) -> AlertRunResult: reasoning_step_events=all_reasoning_step_events, exit_reason=exit_reason, turns_used=turns_used, + run_latency_s=time.monotonic() - run_started, ) finally: if alert_id_to_delete: @@ -953,28 +959,30 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.eval - detail = { - "alert_created": ev.alert_created, - "operator_correct": ev.operator_correct, - "threshold_correct": ev.threshold_correct, - "trigger_correct": ev.trigger_correct, - "filters_correct": ev.filters_correct, - "metric_correct": ev.metric_correct, - "recipients_correct": ev.recipients_correct, - "attributes_correct": ev.attributes_correct, - "granularity_correct": ev.granularity_correct, - "actual_alert_arguments": best.actual_alert_arguments, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + alert_created=ev.alert_created, + operator_correct=ev.operator_correct, + threshold_correct=ev.threshold_correct, + trigger_correct=ev.trigger_correct, + filters_correct=ev.filters_correct, + metric_correct=ev.metric_correct, + recipients_correct=ev.recipients_correct, + attributes_correct=ev.attributes_correct, + granularity_correct=ev.granularity_correct, + actual_alert_arguments=best.actual_alert_arguments, # Why the loop stopped. alert_created=False alone cannot tell a refusal from a run # that hit max_iterations while still on track -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) - exc = AlertSkillAssertionError( + raise_agentic_failure( + AlertSkillAssertionError, f"Alert skill assertion failed. {gate_note} strict_pass={ev.strict_pass}. " f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, " f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, " @@ -982,20 +990,21 @@ def _write_scores(ctx: RunTraceContext) -> None: f"recipients_correct={ev.recipients_correct}, " f"attributes_correct={ev.attributes_correct}, " f"granularity_correct={ev.granularity_correct}. " - f"Actual args: {best.actual_alert_arguments}" + f"Actual args: {best.actual_alert_arguments}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py index 97cb215fc..60441110e 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py @@ -28,6 +28,7 @@ import logging import os import re +import time from dataclasses import dataclass, field from typing import Any @@ -39,6 +40,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -55,9 +57,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -282,6 +284,14 @@ class AnomalyRunResult: response_id: str | None = None tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from + # turn_wall_clock_sec above, which is only the final triggering turn. + run_latency_s: float = 0.0 + # Why the simulated-user loop stopped -- see LoopExit. `triggered=False` alone cannot + # separate a refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -366,6 +376,7 @@ def run_agentic_anomaly_detection( ) def _run_once(conv_id: str) -> AnomalyRunResult: + run_started = time.monotonic() viz_args: dict | None = None execute_result: dict | None = None turn_wall_clock_sec: float | None = None @@ -391,6 +402,10 @@ def _accumulate(result: ChatResult) -> None: all_tool_call_events.extend(result.tool_call_events or []) all_reasoning_step_events.extend(result.reasoning_step_events or []) + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED + for iteration in range(max_iterations): try: chat_result = client.send_message(conv_id, current_question) @@ -403,6 +418,7 @@ def _accumulate(result: ChatResult) -> None: _accumulate(partial) viz_args, execute_result = _extract_anomaly_calls(all_tool_call_events) turn_completed = False + exit_reason = LoopExit.CHAT_ERROR break reasoning_steps.extend(chat_result.reasoning_steps or []) response_id = chat_result.response_id or response_id @@ -416,8 +432,10 @@ def _accumulate(result: ChatResult) -> None: if execute_result is not None: # The turn that ran the detection, not an earlier disambiguation turn. turn_wall_clock_sec = chat_result.turn_wall_clock_sec + exit_reason = LoopExit.SUCCESS break if not response_text: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -426,6 +444,7 @@ def _accumulate(result: ChatResult) -> None: disambiguated = True except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run _log.warning("Simulated anomaly user reply failed for conversation %s: %s", conv_id, exc) + exit_reason = LoopExit.SIMULATED_USER_FAILED break return AnomalyRunResult( @@ -434,10 +453,12 @@ def _accumulate(result: ChatResult) -> None: actual_visualization=viz_args, actual_execute_result=execute_result, turn_wall_clock_sec=turn_wall_clock_sec, + exit_reason=exit_reason, reasoning_steps=reasoning_steps, response_id=response_id, tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, + run_latency_s=time.monotonic() - run_started, ) try: @@ -486,25 +507,29 @@ class AnomalyDetectionAssertionError(AgenticAssertionError): def _detail(best: AnomalyRunResult) -> dict[str, Any]: ev = best.evaluation - return { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "metric_correct": ev.metric_correct, - "granularity_correct": ev.granularity_correct, + return agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + metric_correct=ev.metric_correct, + granularity_correct=ev.granularity_correct, # Which content checks the fixture pinned -- without it a run that verified nothing # reads the same as one where everything matched. - "asserted": ev.asserted, - "disambiguated": ev.disambiguated, - "actual_metrics": sorted(_metric_uris(best.actual_visualization)), - "actual_granularity": _inferred_granularity(best.actual_visualization), + asserted=ev.asserted, + disambiguated=ev.disambiguated, + actual_metrics=sorted(_metric_uris(best.actual_visualization)), + actual_granularity=_inferred_granularity(best.actual_visualization), # Reported, never asserted: whether a real series contains anomalies is a property # of the data, so a fixture demanding some would fail on the next warehouse refresh. - "anomaly_point_count": _point_count(best.actual_execute_result), - "actual_execute_result": best.actual_execute_result, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + anomaly_point_count=_point_count(best.actual_execute_result), + actual_execute_result=best.actual_execute_result, + # Why the loop stopped. triggered=False alone cannot tell a refusal from a run + # that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) def evaluate_agentic_anomaly_detection( @@ -631,20 +656,24 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Analysed {detail['actual_metrics']} at {detail['actual_granularity']}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = AnomalyDetectionAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = len(summary.run_results) - raise exc - - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=len(summary.run_results), + raise_agentic_failure( + AnomalyDetectionAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, + ) + + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py index 25e0260d0..8cfe063ed 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py @@ -19,6 +19,7 @@ import json import os import re +import time from dataclasses import dataclass, field from datetime import date from typing import ClassVar, Literal @@ -40,6 +41,7 @@ summarize_visualizations, ) from gooddata_eval.core.agentic._gate import log_gate_scores +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -65,7 +67,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.scoring import ( check_filters, @@ -611,6 +612,11 @@ class ConversationResult: stalled_turns: int = 0 mode: ConversationMode = "legacy" judge_model: str | None = None + # This conversation's own wall time, across every turn -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). There is no + # K-of-N selection for this kind (see run_agentic_conversation's docstring), so + # this is simply the one conversation's own time, not a "best of" pick. + run_latency_s: float = 0.0 def _metric_creations(tool_call_events: list[ToolCallEvent]) -> list[tuple[str, bool]]: @@ -729,6 +735,7 @@ def run_agentic_conversation( ``user_context`` is the context attached when the conversation starts; a turn's own ``user_context`` replaces it from that turn on. """ + run_started = time.monotonic() resolved_mode = resolve_conversation_mode(mode) if fresh_conversation_per_turn and initial_conversation_id is not None: raise ValueError("fresh_conversation_per_turn needs conversations this run creates itself") @@ -1038,25 +1045,27 @@ def run_agentic_conversation( stalled_turns=stalled, mode=resolved_mode, judge_model=judge.model_name if judged else None, + run_latency_s=time.monotonic() - run_started, ) def _conversation_detail(result: ConversationResult) -> dict: - return { - "mode": result.mode, - "full_skill_coverage": result.full_skill_coverage, - "conversation_success": result.conversation_success, - "context_success": result.context_success, - "context_kept_rate": result.context_kept_rate, - "turns_before_first_break": result.turns_before_first_break, - "lost_context_clarifications": result.lost_context_clarifications, - "stalled_turns": result.stalled_turns, - "total_clarification_turns": result.total_clarification_turns, - "max_clarification_turns": result.max_clarification_turns, - "judge_model": result.judge_model, - "turns": [tr.detail() for tr in result.turn_results], - **timeline_detail(result.tool_call_events, result.reasoning_step_events), - } + return agentic_detail( + result.tool_call_events, + result.reasoning_step_events, + mode=result.mode, + full_skill_coverage=result.full_skill_coverage, + conversation_success=result.conversation_success, + context_success=result.context_success, + context_kept_rate=result.context_kept_rate, + turns_before_first_break=result.turns_before_first_break, + lost_context_clarifications=result.lost_context_clarifications, + stalled_turns=result.stalled_turns, + total_clarification_turns=result.total_clarification_turns, + max_clarification_turns=result.max_clarification_turns, + judge_model=result.judge_model, + turns=[tr.detail() for tr in result.turn_results], + ) class ConversationAssertionError(AgenticAssertionError): @@ -1236,21 +1245,25 @@ def _write_scores(ctx: RunTraceContext) -> None: f"full_skill_coverage={result.full_skill_coverage}. " f"Failed turns: {legacy_failed}" ) - exc = ConversationAssertionError(message) - exc.reasoning_steps = result.reasoning_steps - exc.conversation_id = result.conversation_id - exc.response_id = result.response_id - exc.detail = detail # This kind takes no k and drives its fixture exactly once, whatever --runs asks # for. Saying so explicitly stops the report claiming K runs that never happened. - exc.runs_passed = 0 - exc.runs_effective = 1 - raise exc - return AgenticEvalOutcome( + raise_agentic_failure( + ConversationAssertionError, + message, + reasoning_steps=result.reasoning_steps, + conversation_id=result.conversation_id, + response_id=result.response_id, + detail=detail, + runs_passed=0, + runs_effective=1, + best_run_latency_s=result.run_latency_s, + ) + return agentic_success( reasoning_steps=result.reasoning_steps, conversation_id=result.conversation_id, response_id=result.response_id, detail=detail, runs_passed=1, runs_effective=1, + best_run_latency_s=result.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py index 1496173b4..02e5483b2 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py @@ -17,6 +17,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -27,18 +28,18 @@ utc_now, ) from gooddata_eval.core.chat.render import render_answer_text -from gooddata_eval.core.chat.sse_client import ChatClient +from gooddata_eval.core.chat.sse_client import ChatClient, ChatError from gooddata_eval.core.config import ReasoningEffort from gooddata_eval.core.models import ( AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) -from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings +from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings _DEFAULT_K = 1 # Matches kda_skill and visualization. The reply this loop sends is built from the expectation @@ -664,6 +665,12 @@ class DashboardRunResult: tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) timings: PhaseTimings = field(default_factory=PhaseTimings) + # Why the loop stopped -- see LoopExit. A dashboard not produced alone cannot tell a + # refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). CHAT_ERROR is reachable; SIMULATED_USER_FAILED is not -- + # this loop's follow-up is built from the expectation by build_simulated_reply, with no + # judge call that could fail. + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -864,11 +871,42 @@ def _execute_single_dashboard_run( turn_offset = 0.0 tool_index_offset = 0 reasoning_index_offset = 0 + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() (or the is_edit "no reply of its own" branch below, which + # is the same shape of outcome -- the agent answered, just not with the dashboard tool) + # is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED for iteration in range(max_iterations): turns += 1 agent_started = time.monotonic() - chat_result = client.send_message(conversation_id, current_question) + try: + chat_result = client.send_message(conversation_id, current_question) + except ChatError as exc: + # Recorded rather than left to escape: this runs inside one of K runs, so an + # uncaught ChatError discarded every run already completed along with this one's + # exit_reason, and the item surfaced as a bare error with no per-run diagnosis. + # Matches metric_skill.py and conversation.py, which record it the same way. + timings.agent_s += time.monotonic() - agent_started + print(f"[CHAT] send_message failed for conversation {conversation_id}: {exc}") + partial = exc.partial_result + if partial is not None: + # Shifted and indexed exactly as a completed turn is, before anything reads + # the events: left raw, a late turn's call_ts restarts near zero and + # timeline_detail reports it as overlapping the first one. + reasoning_steps.extend(partial.reasoning_steps or []) + response_id = partial.response_id or response_id + turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events( + partial, + turn_offset=turn_offset, + tool_index_offset=tool_index_offset, + reasoning_index_offset=reasoning_index_offset, + ) + all_tool_call_events.extend(partial.tool_call_events or []) + all_reasoning_step_events.extend(partial.reasoning_step_events or []) + steps += partial.reasoning_step_count + exit_reason = LoopExit.CHAT_ERROR + break agent_elapsed = time.monotonic() - agent_started timings.agent_s += agent_elapsed reasoning_steps.extend(chat_result.reasoning_steps or []) @@ -903,10 +941,12 @@ def _execute_single_dashboard_run( dashboard_part = _extract_dashboard_part(chat_result, "dashboard") if is_edit: patch_part = _extract_dashboard_part(chat_result, _PATCH_TYPE) + exit_reason = LoopExit.SUCCESS break response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) if not response_text and not chat_result.tool_call_events: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -941,6 +981,7 @@ def _execute_single_dashboard_run( tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, timings=timings, + exit_reason=exit_reason, ) @@ -1108,13 +1149,17 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail: dict[str, Any] = { + detail: dict[str, Any] = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, **best.evaluation.strict_checks, **best.evaluation.diagnostics, - "failures": best.evaluation.failures, - "notes": best.evaluation.notes, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + failures=best.evaluation.failures, + notes=best.evaluation.notes, + # Why the loop stopped. A dashboard not produced alone cannot tell a refusal from + # a run that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -1135,26 +1180,28 @@ def _write_scores(ctx: RunTraceContext) -> None: " one feature flag registers both, so check that before the model." ) notes = "; ".join(best.evaluation.notes) - exc = DashboardSkillAssertionError( + raise_agentic_failure( + DashboardSkillAssertionError, f"Dashboard skill assertion failed. {gate_note}{skill_note} " f"Checks: {best.evaluation.strict_checks}. " f"Failures: {'; '.join(best.evaluation.failures) or 'none reported'}." - + (f" Notes (not scored): {notes}." if notes else "") + + (f" Notes (not scored): {notes}." if notes else ""), + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py index dfbbc9e2c..44f427266 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py @@ -14,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -32,9 +33,8 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, ) -from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings +from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings _DEFAULT_K = 1 @@ -318,42 +318,46 @@ def _write_scores(ctx: RunTraceContext) -> None: + " | ".join(unscored) ) exc.timings = item_timings + exc.best_run_latency_s = run_latency_s(summary.best.timings) raise exc runs_passed = sum(1 for r in summary.scored_run_results if r.passed) runs_effective = len(summary.run_results) best = summary.best - detail = { - "judge_passed": best.passed, - "judge_reasoning": best.reasoning, - "actual_output": best.actual_output, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + judge_passed=best.passed, + judge_reasoning=best.reasoning, + actual_output=best.actual_output, # Only present when it happened, so the usual JSON shape is unchanged. A # pass@K computed over fewer runs than --runs asked for is a weaker result and # the report has to say so. **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), - } + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - exc = GeneralQuestionAssertionError( - f"General question assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}" + raise_agentic_failure( + GeneralQuestionAssertionError, + f"General question assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py index 790992027..f61d1da37 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py @@ -3,6 +3,7 @@ from __future__ import annotations +import time from dataclasses import dataclass, field from gooddata_eval.core.agentic._gate import ( @@ -13,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -32,7 +34,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - timeline_detail, ) _DEFAULT_K = 1 @@ -83,6 +84,9 @@ class GuardrailResult: # Set when the judge returned something unreadable for THIS run. Excluded from pass@K # and from Langfuse scoring rather than counted as a failure -- see score_run. judge_error: str | None = None + # This run's own wall time (agent call + its grading) -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -117,6 +121,7 @@ def _run_single_guardrail( Extracted so the two call sites below (the first conversation, which may be supplied, and the remaining K-1) cannot drift -- they had already duplicated the whole body once. """ + run_started = time.monotonic() chat_result = client.send_message(conversation_id, question) actual_output = render_answer_text(chat_result) verdict = score_run(judge, input=question, expected_output=expected_output, actual_output=actual_output) @@ -131,6 +136,7 @@ def _run_single_guardrail( tool_call_events=list(chat_result.tool_call_events or []), reasoning_step_events=list(chat_result.reasoning_step_events or []), judge_error=verdict.error, + run_latency_s=time.monotonic() - run_started, ) @@ -292,42 +298,47 @@ def _write_scores(ctx: RunTraceContext) -> None: if not summary.scored_run_results: # No readable verdict for any run: an error, not K failures. Raised after the # trace link is queued so whatever the agent did is still linked. - raise JudgeResponseError( + exc = JudgeResponseError( f"judge returned no readable verdict for any of the {len(summary.run_results)} run(s): " + " | ".join(unscored) ) + exc.best_run_latency_s = summary.best.run_latency_s + raise exc runs_passed = sum(1 for r in summary.scored_run_results if r.passed) runs_effective = len(summary.run_results) best = summary.best - detail = { - "judge_passed": best.passed, - "judge_reasoning": best.reasoning, - "actual_output": best.actual_output, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + judge_passed=best.passed, + judge_reasoning=best.reasoning, + actual_output=best.actual_output, # Only present when it happened, so the usual JSON shape is unchanged. A # pass@K over fewer runs than --runs asked for is a weaker result. **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), - } + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - exc = GuardrailAssertionError( - f"Guardrail assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}" + raise_agentic_failure( + GuardrailAssertionError, + f"Guardrail assertion failed. {gate_note} passed={best.passed}. Reasoning: {best.reasoning}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py index de87f2c5b..aabc1e9dc 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py @@ -5,6 +5,7 @@ import logging import os +import time from dataclasses import dataclass, field import httpx @@ -17,6 +18,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -36,7 +38,6 @@ LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -198,6 +199,10 @@ class KdaRunResult: # separate a refusal from a run that hit max_iterations while still on track. exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED turns_used: int = 0 + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from + # turn_wall_clock_sec above, which is only the final create-triggering turn. + run_latency_s: float = 0.0 @dataclass @@ -266,6 +271,7 @@ def run_agentic_kda_skill( ) def _run_once(conv_id: str) -> KdaRunResult: + run_started = time.monotonic() create_args: dict | None = None execute_result: dict | None = None turn_wall_clock_sec: float | None = None @@ -373,6 +379,7 @@ def _accumulate(result: ChatResult) -> None: reasoning_step_events=all_reasoning_step_events, exit_reason=exit_reason, turns_used=turns_used, + run_latency_s=time.monotonic() - run_started, ) try: @@ -529,20 +536,21 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.evaluation - detail = { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "disambiguated": ev.disambiguated, - "actual_create_args": best.actual_create_args, - "actual_execute_result": best.actual_execute_result, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + disambiguated=ev.disambiguated, + actual_create_args=best.actual_create_args, + actual_execute_result=best.actual_execute_result, # Why the loop stopped -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -553,19 +561,23 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Actual create args: {best.actual_create_args}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = KdaSkillAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + raise_agentic_failure( + KdaSkillAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, + ) + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py index 04ccdc032..24a796669 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py @@ -20,6 +20,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -40,9 +41,8 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) -from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings +from gooddata_eval.core.timing import PhaseTimings, log_timer, run_latency_s, sum_timings try: from openai import OpenAI as _OpenAI @@ -603,42 +603,45 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best expected_outputs_list: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output] - detail = { - "metric_created": best.metric_created, - "maql_correct": best.maql_correct, - "expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list], - "actual_maql": best.actual_maql, + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + metric_created=best.metric_created, + maql_correct=best.maql_correct, + expected_maql_candidates=[c.get("maql", "") for c in expected_outputs_list], + actual_maql=best.actual_maql, # Why the loop stopped. metric_created=False alone cannot tell a refusal from a run # that hit max_iterations while still on track -- see LoopExit. - "exit_reason": best.exit_reason.value, - "turns_used": best.turns_used, - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=best.turns_used, + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) candidates_str = "; ".join(repr(c.get("maql", "")) for c in expected_outputs_list) - exc = MetricSkillAssertionError( + raise_agentic_failure( + MetricSkillAssertionError, f"Metric skill assertion failed. {gate_note} " f"metric_created={best.metric_created}, maql_correct={best.maql_correct}. " f"Expected MAQL (candidates): {candidates_str}. " - f"Actual MAQL: {best.actual_maql}." + f"Actual MAQL: {best.actual_maql}.", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), + timings=item_timings, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=run_latency_s(best.timings), timings=item_timings, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py index 9194922c0..60326da8c 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py @@ -3,6 +3,7 @@ from __future__ import annotations +import time from dataclasses import dataclass, field from gooddata_eval.core.agentic._gate import ( @@ -13,6 +14,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -29,7 +31,6 @@ AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, ) _DEFAULT_K = 1 @@ -75,6 +76,9 @@ class SearchResult: response_id: str | None = None tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + # This run's own wall time -- mirrors the single-shot path's best_run_latency_s + # (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -113,6 +117,7 @@ def run_agentic_search_tool( try: conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() try: + run_started = time.monotonic() chat_result = client.send_message(conv_id_0, question) tcs = chat_result.tool_call_events or [] selected = _tool_selection(tcs) @@ -127,6 +132,7 @@ def run_agentic_search_tool( response_id=chat_result.response_id, tool_call_events=list(chat_result.tool_call_events or []), reasoning_step_events=list(chat_result.reasoning_step_events or []), + run_latency_s=time.monotonic() - run_started, ) ) finally: @@ -136,6 +142,7 @@ def run_agentic_search_tool( for _ in range(1, k): conv_id = client.create_conversation() try: + run_started = time.monotonic() chat_result = client.send_message(conv_id, question) tcs = chat_result.tool_call_events or [] selected = _tool_selection(tcs) @@ -150,6 +157,7 @@ def run_agentic_search_tool( response_id=chat_result.response_id, tool_call_events=list(chat_result.tool_call_events or []), reasoning_step_events=list(chat_result.reasoning_step_events or []), + run_latency_s=time.monotonic() - run_started, ) ) finally: @@ -263,32 +271,35 @@ def _write_scores(ctx: RunTraceContext) -> None: runs_effective = len(summary.run_results) best = summary.best - detail = { - "tool_selected": best.tool_selected, - "tool_correct": best.tool_correct, - "tool_call_names": best.tool_call_names, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + tool_selected=best.tool_selected, + tool_correct=best.tool_correct, + tool_call_names=best.tool_call_names, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) - exc = SearchToolAssertionError( + raise_agentic_failure( + SearchToolAssertionError, f"Search tool assertion failed. {gate_note} " f"tool_selected={best.tool_selected}, tool_correct={best.tool_correct}. " - f"Tool calls made: {best.tool_call_names}" + f"Tool calls made: {best.tool_call_names}", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py index dfd5c5c65..976a52707 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py @@ -8,6 +8,7 @@ from __future__ import annotations import os +import time from dataclasses import dataclass, field from gooddata_eval.core.agentic._gate import ( @@ -18,6 +19,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -46,7 +48,6 @@ ReasoningStepEvent, ToolCallEvent, shift_and_index_events, - timeline_detail, ) from gooddata_eval.core.scoring import get_dimension_uri_set, get_metric_uri_set, uri_to_display_name @@ -71,6 +72,9 @@ class RunResult: # Why the simulated-user loop stopped. visualization_created=False alone cannot separate # a refusal from a run that hit max_iterations while still on track -- see LoopExit. exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). + run_latency_s: float = 0.0 @dataclass @@ -193,6 +197,7 @@ def _execute_single_run( requires_execution: bool = False, ) -> RunResult: """Drive one full multi-turn conversation and evaluate the result.""" + run_started = time.monotonic() total_turns = 0 total_steps = 0 all_tool_call_events: list[ToolCallEvent] = [] @@ -296,6 +301,7 @@ def _execute_single_run( tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, exit_reason=exit_reason, + run_latency_s=time.monotonic() - run_started, ) @@ -511,15 +517,16 @@ def _write_scores(ctx: RunTraceContext) -> None: best = summary.best ev = best.eval_result - detail = { + detail = agentic_detail( + best.tool_call_events, + best.reasoning_step_events, **evaluation_result_detail(ev), # Why the loop stopped -- see LoopExit. total_turns is already the turn count for # this run, so it doubles as turns_used. - "exit_reason": best.exit_reason.value, - "turns_used": int(best.total_turns), - "max_iterations": max_iterations, - **timeline_detail(best.tool_call_events, best.reasoning_step_events), - } + exit_reason=best.exit_reason.value, + turns_used=int(best.total_turns), + max_iterations=max_iterations, + ) if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective) @@ -528,7 +535,8 @@ def _write_scores(ctx: RunTraceContext) -> None: cross_ref_detail = (" → " + "; ".join(ev.cross_ref_errors)) if ev.cross_ref_errors else "" expected_dump = best.best_expected.model_dump(exclude_none=True) actual_dump = best.actual_output.model_dump(exclude_none=True) if best.actual_output else None - exc = VisualizationAssertionError( + raise_agentic_failure( + VisualizationAssertionError, "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" "Agentic Visualization Assertion Failed! (Critical Mode)\n" f"{gate_note}\n" @@ -558,20 +566,21 @@ def _write_scores(ctx: RunTraceContext) -> None: f" Viz Type Hard : {ev.viz_type_hard}\n" f" Executed : {ev.executed}{' (required)' if ev.requires_execution else ''}\n" f" Stated Value Matches : {ev.stated_value_matches} (reported only)\n" - "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" + "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n", + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=runs_effective, + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py index 5c84a1197..e26f24e04 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py @@ -22,6 +22,7 @@ import logging import os +import time from dataclasses import dataclass, field from typing import Any @@ -33,6 +34,7 @@ log_gate_scores, stamp_gate_metadata, ) +from gooddata_eval.core.agentic._outcome import agentic_detail, agentic_success, raise_agentic_failure from gooddata_eval.core.agentic._trace_linker import ( RunIdentity, RunTraceContext, @@ -50,9 +52,9 @@ AgenticAssertionError, AgenticEvalOutcome, ChatResult, + LoopExit, ReasoningStepEvent, ToolCallEvent, - build_latency_breakdown, shift_and_index_events, ) @@ -209,6 +211,14 @@ class WhatIfRunResult: response_id: str | None = None tool_call_events: list[ToolCallEvent] = field(default_factory=list) reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + # This run's own wall time, across all its turns -- mirrors the single-shot + # path's best_run_latency_s (see core/runner.py's _run_one_item). Distinct from + # turn_wall_clock_sec above, which is only the scenario-execution turn. + run_latency_s: float = 0.0 + # Why the simulated-user loop stopped -- see LoopExit. `triggered=False` alone cannot + # separate a refusal from a run that hit max_iterations while still on track (matches + # kda_skill.py/alert_skill.py). + exit_reason: LoopExit = LoopExit.BUDGET_EXHAUSTED @dataclass @@ -323,6 +333,7 @@ def run_agentic_what_if( ) def _run_once(conv_id: str) -> WhatIfRunResult: + run_started = time.monotonic() create_args: dict | None = None execute_result: dict | None = None turn_wall_clock_sec: float | None = None @@ -348,6 +359,10 @@ def _accumulate(result: ChatResult) -> None: all_tool_call_events.extend(result.tool_call_events or []) all_reasoning_step_events.extend(result.reasoning_step_events or []) + # Defaults to BUDGET_EXHAUSTED: every other exit assigns explicitly, so a loop that + # simply runs out of range() is labelled correctly with no trailing else. + exit_reason = LoopExit.BUDGET_EXHAUSTED + for iteration in range(max_iterations): try: chat_result = client.send_message(conv_id, current_question) @@ -360,6 +375,7 @@ def _accumulate(result: ChatResult) -> None: _accumulate(partial) create_args, execute_result = _extract_what_if_calls(all_tool_call_events) turn_completed = False + exit_reason = LoopExit.CHAT_ERROR break reasoning_steps.extend(chat_result.reasoning_steps or []) response_id = chat_result.response_id or response_id @@ -373,8 +389,10 @@ def _accumulate(result: ChatResult) -> None: if execute_result is not None: # The turn that ran the scenario, not an earlier disambiguation turn. turn_wall_clock_sec = chat_result.turn_wall_clock_sec + exit_reason = LoopExit.SUCCESS break if not response_text: + exit_reason = LoopExit.AGENT_SILENT break if iteration >= max_iterations - 1: break @@ -383,6 +401,7 @@ def _accumulate(result: ChatResult) -> None: disambiguated = True except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run _log.warning("Simulated what-if user reply failed for conversation %s: %s", conv_id, exc) + exit_reason = LoopExit.SIMULATED_USER_FAILED break return WhatIfRunResult( @@ -390,11 +409,13 @@ def _accumulate(result: ChatResult) -> None: evaluation=_evaluate_run(create_args, execute_result, expected_output, turn_completed, disambiguated), actual_create_args=create_args, actual_execute_result=execute_result, + exit_reason=exit_reason, turn_wall_clock_sec=turn_wall_clock_sec, reasoning_steps=reasoning_steps, response_id=response_id, tool_call_events=all_tool_call_events, reasoning_step_events=all_reasoning_step_events, + run_latency_s=time.monotonic() - run_started, ) try: @@ -446,26 +467,30 @@ class WhatIfAssertionError(AgenticAssertionError): def _detail(best: WhatIfRunResult) -> dict[str, Any]: ev = best.evaluation adjustments = _adjustments(best.actual_create_args) - return { - "triggered": ev.triggered, - "executed": ev.executed, - "success": ev.success, - "turn_completed": ev.turn_completed, - "metric_correct": ev.metric_correct, - "maql_correct": ev.maql_correct, - "scenario_count_correct": ev.scenario_count_correct, - "baseline_correct": ev.baseline_correct, + return agentic_detail( + best.tool_call_events, + best.reasoning_step_events, + triggered=ev.triggered, + executed=ev.executed, + success=ev.success, + turn_completed=ev.turn_completed, + metric_correct=ev.metric_correct, + maql_correct=ev.maql_correct, + scenario_count_correct=ev.scenario_count_correct, + baseline_correct=ev.baseline_correct, # Which content checks the fixture pinned -- without it a run that verified nothing # reads the same as one where everything matched. - "asserted": ev.asserted, - "disambiguated": ev.disambiguated, - "actual_adjustments": adjustments, - "actual_scenario_labels": [ + asserted=ev.asserted, + disambiguated=ev.disambiguated, + actual_adjustments=adjustments, + actual_scenario_labels=[ s.get("label") for s in ((best.actual_create_args or {}).get("scenarios") or []) if isinstance(s, dict) ], - "actual_execute_result": best.actual_execute_result, - "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), - } + actual_execute_result=best.actual_execute_result, + # Why the loop stopped. triggered=False alone cannot tell a refusal from a run + # that hit max_iterations while still on track -- see LoopExit. + exit_reason=best.exit_reason.value, + ) def evaluate_agentic_what_if( @@ -595,20 +620,24 @@ def _write_scores(ctx: RunTraceContext) -> None: f"Actual adjustments: {detail['actual_adjustments']}. " f"Actual execute result: {best.actual_execute_result}." ) - exc = WhatIfAssertionError(message) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = len(summary.run_results) - raise exc - - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=len(summary.run_results), + raise_agentic_failure( + WhatIfAssertionError, + message, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, + ) + + return agentic_success( reasoning_steps=best.reasoning_steps, conversation_id=best.conversation_id, response_id=best.response_id, detail=detail, + runs_passed=runs_passed, + runs_effective=len(summary.run_results), + best_run_latency_s=best.run_latency_s, ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py b/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py index d7e9ddc5f..4c9ee2dca 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/evaluators/_llm_judge.py @@ -49,6 +49,7 @@ class JudgeResponseError(RuntimeError): # rather than attached loosely, because the runner reads it off the exception to report # what an unevaluable item still cost. timings: PhaseTimings + best_run_latency_s: float | None def _message_content(response: Any) -> str | None: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/models.py b/packages/gooddata-eval/src/gooddata_eval/core/models.py index 142fc20f6..79067e42e 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/models.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/models.py @@ -408,6 +408,7 @@ class AgenticAssertionError(AssertionError): timings: PhaseTimings runs_passed: int runs_effective: int + best_run_latency_s: float | None class AgenticEvalOutcome(BaseModel): @@ -433,6 +434,12 @@ class AgenticEvalOutcome(BaseModel): # ran -- agentic_conversation drives its fixture once whatever --runs says. runs_passed: int = 0 runs_effective: int = 0 + # The rank-selected run's own wall time -- mirrors the single-shot path's + # ItemReport.best_run_latency_s (see core/runner.py's _run_one_item), which the + # agentic path never populated because its K runs happen inside the evaluator, + # not in a loop the runner itself can time. None for a kind that has not yet been + # wired to measure it (see AgenticAssertionError.best_run_latency_s, same default). + best_run_latency_s: float | None = None class SummaryInput(BaseModel): diff --git a/packages/gooddata-eval/src/gooddata_eval/core/timing.py b/packages/gooddata-eval/src/gooddata_eval/core/timing.py index a3c05aef1..f9d6e2812 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/timing.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/timing.py @@ -72,3 +72,13 @@ def as_dict(self) -> dict[str, float]: def sum_timings(timings: list[PhaseTimings]) -> PhaseTimings: """Total across a list of per-run timings (empty list -> all zeroes).""" return sum(timings, PhaseTimings()) + + +def run_latency_s(timings: PhaseTimings) -> float: + """One run's own wall time, for a single run's ``PhaseTimings``. + + Excludes ``langfuse_s`` (deliberately off the critical path -- see + ``PhaseTimings.langfuse_s``) so this stays comparable to the single-shot path's + ``best_run_latency_s``, which times only the agent call and its grading. + """ + return timings.agent_s + timings.judge_s + timings.simulated_user_s diff --git a/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py b/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py index 84e138fb1..f3aa69534 100644 --- a/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py +++ b/packages/gooddata-eval/tests/test_agentic_anomaly_detection.py @@ -16,7 +16,7 @@ evaluate_agentic_anomaly_detection, run_agentic_anomaly_detection, ) -from gooddata_eval.core.models import ChatResult +from gooddata_eval.core.models import ChatResult, LoopExit _MODULE = "gooddata_eval.core.agentic.anomaly_detection" @@ -254,18 +254,22 @@ def test_a_single_turn_detection_passes(): summary = _run([_chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is False + assert summary.best.exit_reason is LoopExit.SUCCESS def test_a_clarifying_question_is_answered_and_the_run_continues(): summary = _run([_chat([], text="Which Spend metric did you mean?"), _chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is True + assert summary.best.exit_reason is LoopExit.SUCCESS def test_the_loop_stops_at_max_iterations_without_a_detection(): + # executed=False alone cannot tell this apart from a refusal -- exit_reason can. summary = _run([_chat([], text="Still thinking")] * 4, max_iterations=4) assert summary.pass_at_k is False assert summary.best.evaluation.executed is False + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_response_ends_the_run_immediately(): @@ -276,17 +280,19 @@ def test_an_empty_response_ends_the_run_immediately(): patch(f"{_MODULE}.ChatClient", return_value=client), patch(f"{_MODULE}.generate_simulated_anomaly_response") as sim, ): - run_agentic_anomaly_detection( + summary = run_agentic_anomaly_detection( host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED ) assert client.send_message.call_count == 1 sim.assert_not_called() + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_chat_error_on_a_later_run_does_not_discard_the_earlier_one(): summary = _run([_chat(_pair()), RuntimeError("boom")], k=2) assert len(summary.run_results) == 2 assert summary.pass_at_k is True + assert summary.run_results[1].exit_reason is LoopExit.CHAT_ERROR def test_k_must_be_at_least_one(): @@ -318,6 +324,8 @@ def test_detail_reports_the_series_analysed_and_the_flag_count(): assert outcome.detail["anomaly_point_count"] == 0 assert outcome.detail["asserted"] == ["metric", "granularity"] assert "latency_breakdown" in outcome.detail + assert "tool_calls" in outcome.detail + assert outcome.detail["exit_reason"] == "success" assert outcome.runs_passed == 1 diff --git a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py index e949d8a3b..23b26a7b2 100644 --- a/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_dashboard_skill.py @@ -15,7 +15,8 @@ evaluate_dashboard_response, run_agentic_dashboard_skill, ) -from gooddata_eval.core.models import ChatResult, ToolCallEvent +from gooddata_eval.core.chat.sse_client import ChatError +from gooddata_eval.core.models import ChatResult, LoopExit, ToolCallEvent # Ids and titles are the ones the eval layout seeds; shapes are trimmed from real runs of # `agent_dashboard_skill` against ecommerce_demo. @@ -1086,6 +1087,8 @@ def test_the_loop_is_capped(self): assert client.send_message.call_count == 2 assert not summary.pass_at_k assert not summary.best.evaluation.drafted + # drafted=False alone cannot tell this apart from a refusal -- exit_reason can. + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_turn_stops_the_loop(self): client = MagicMock() @@ -1093,6 +1096,7 @@ def test_an_empty_turn_stops_the_loop(self): summary = _run_with(client, _DC05_EXPECTED, max_iterations=5) assert client.send_message.call_count == 1 assert not summary.pass_at_k + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_caller_supplied_conversation_is_not_deleted(self): client = MagicMock() @@ -1101,6 +1105,54 @@ def test_a_caller_supplied_conversation_is_not_deleted(self): client.create_conversation.assert_not_called() client.delete_conversation.assert_not_called() + def test_a_chat_error_is_recorded_rather_than_left_to_escape(self): + """It used to escape run_agentic_dashboard_skill, taking every completed K-run with + it: no DashboardRunResult, no exit_reason, just a bare error on the item. Recorded + the way metric_skill.py and conversation.py record it, so an infrastructure fault + stays distinguishable from the agent failing to draft.""" + client = MagicMock() + client.send_message.side_effect = ChatError("gen-ai fell over") + + summary = _run_with(client, _DC05_EXPECTED) + + assert summary.run_results[0].exit_reason is LoopExit.CHAT_ERROR + assert summary.run_results[0].evaluation.strict_pass is False + assert summary.pass_at_k is False + + def test_a_chat_error_keeps_the_partial_turn_s_work(self): + """A bare catch would discard ChatError.partial_result. What the stream delivered + before it died is still this turn's work -- the steps were taken and the tools ran.""" + partial = _chat_result( + tool_calls=[{"functionName": "search_objects", "functionArguments": "{}", "callTs": 1, "durationMs": 10}], + text="Here is the draft.", + ) + partial.reasoning_steps = ["looked for the charts"] + partial.reasoning_step_count = 1 + client = MagicMock() + client.send_message.side_effect = ChatError("stream died mid-answer", partial_result=partial) + + summary = _run_with(client, _DC05_EXPECTED) + run = summary.run_results[0] + + assert run.exit_reason is LoopExit.CHAT_ERROR + assert run.reasoning_steps == ["looked for the charts"] + assert run.total_steps == 1 + assert [tc.function_name for tc in run.tool_call_events] == ["search_objects"] + + def test_a_chat_error_on_a_later_turn_keeps_the_earlier_ones(self): + """The whole point of recording it: turns already completed survive the fault.""" + client = MagicMock() + client.send_message.side_effect = [ + _chat_result(text="Which charts did you have in mind?"), + ChatError("gen-ai fell over"), + ] + + summary = _run_with(client, _DC05_EXPECTED) + run = summary.run_results[0] + + assert run.exit_reason is LoopExit.CHAT_ERROR + assert run.total_turns == 2 + class TestEvaluateEntryPoint: def test_a_failing_run_raises_with_the_failures_attached(self): @@ -1161,6 +1213,8 @@ def test_a_passing_run_returns_the_outcome(self): assert outcome.runs_passed == 1 assert outcome.runs_effective == 1 assert outcome.detail["charts_matched"] is True + assert outcome.detail["exit_reason"] == "success" + assert "tool_calls" in outcome.detail def _as_events(raw: list[dict]) -> list[ToolCallEvent]: diff --git a/packages/gooddata-eval/tests/test_agentic_general_question.py b/packages/gooddata-eval/tests/test_agentic_general_question.py index 3d265b00a..7f5f2b131 100644 --- a/packages/gooddata-eval/tests/test_agentic_general_question.py +++ b/packages/gooddata-eval/tests/test_agentic_general_question.py @@ -257,6 +257,7 @@ def test_evaluate_agentic_general_question_returns_reasoning_steps_on_pass(): "judge_reasoning": "Correct answer", "actual_output": "42", "latency_breakdown": [], + "tool_calls": [], } @@ -284,6 +285,7 @@ def test_evaluate_agentic_general_question_attaches_reasoning_steps_to_exception "judge_reasoning": "Wrong answer", "actual_output": "I don't know", "latency_breakdown": [], + "tool_calls": [], } @@ -400,6 +402,9 @@ def test_evaluate_general_question_aggregates_timings_across_k_runs(): assert outcome.timings.agent_s == 5.0 # 2.0 + 3.0 assert outcome.timings.judge_s == 1.5 # 1.0 + 0.5 + # best_run_latency_s is the rank-selected run's own time (agent_s=2.0 + judge_s=1.0 + # from the first, tied-best run), not the item total above (both runs summed). + assert outcome.best_run_latency_s == 3.0 def test_evaluate_general_question_attaches_timings_to_the_failure_exception(): @@ -423,6 +428,7 @@ def test_evaluate_general_question_attaches_timings_to_the_failure_exception(): assert exc_info.value.timings.agent_s == 4.0 assert exc_info.value.timings.judge_s == 2.0 + assert exc_info.value.best_run_latency_s == 6.0 def test_trace_link_window_is_pinned_at_submit_time_not_when_the_task_runs(): diff --git a/packages/gooddata-eval/tests/test_agentic_kda_skill.py b/packages/gooddata-eval/tests/test_agentic_kda_skill.py index 573a2a267..3d6f5b620 100644 --- a/packages/gooddata-eval/tests/test_agentic_kda_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_kda_skill.py @@ -1134,6 +1134,7 @@ def test_evaluate_agentic_kda_skill_returns_reasoning_steps_on_pass(): "turns_used": 1, "max_iterations": 1, "latency_breakdown": [], + "tool_calls": [], } @@ -1172,6 +1173,7 @@ def test_evaluate_agentic_kda_skill_attaches_reasoning_steps_to_exception_on_fai "turns_used": 1, "max_iterations": 1, "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_metric_skill.py b/packages/gooddata-eval/tests/test_agentic_metric_skill.py index 37e23f3e2..6de84c550 100644 --- a/packages/gooddata-eval/tests/test_agentic_metric_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_metric_skill.py @@ -811,6 +811,7 @@ def test_evaluate_metric_skill_surfaces_timings_on_the_outcome(): assert outcome.timings.agent_s == 3.0 assert outcome.timings.simulated_user_s == 0.0 + assert outcome.best_run_latency_s == 3.0 def test_no_timer_output_by_default(monkeypatch, capsys): diff --git a/packages/gooddata-eval/tests/test_agentic_runner.py b/packages/gooddata-eval/tests/test_agentic_runner.py index 23586b563..c20402c2c 100644 --- a/packages/gooddata-eval/tests/test_agentic_runner.py +++ b/packages/gooddata-eval/tests/test_agentic_runner.py @@ -344,6 +344,54 @@ def test_run_agentic_items_records_phase_timings_when_the_item_fails(): assert report.items[0].judge_latency_s == 2.0 +def test_run_agentic_items_records_best_run_latency_on_success(): + # best_run_latency_s was never populated on the agentic path -- the K-run loop and + # best-of-K selection happen inside the evaluator, not in a loop the runner itself + # times (unlike _run_one_item on the single-shot path). + outcome = AgenticEvalOutcome( + conversation_id="c1", + detail={"alert_created": True}, + best_run_latency_s=4.25, + ) + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", return_value=outcome): + report = run_agentic_items( + [_timed_item()], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + + assert report.items[0].best_run_latency_s == 4.25 + + +def test_run_agentic_items_records_best_run_latency_when_the_item_fails(): + exc = AlertSkillAssertionError("nope") + exc.detail = {"alert_created": False} + exc.best_run_latency_s = 9.5 + + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", side_effect=exc): + report = run_agentic_items( + [_timed_item()], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + + assert report.items[0].pass_at_k is False + assert report.items[0].best_run_latency_s == 9.5 + + +def test_an_errored_item_without_best_run_latency_keeps_the_none_default(): + # Kinds not yet wired to measure it (or any exception with no such attribute) must + # not report an invented number -- None stays None, not 0.0. + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", side_effect=RuntimeError("boom")): + report = run_agentic_items([_timed_item()], host="h", token="t", workspace_id="ws1", run_ts="2026-01-01") + + assert report.items[0].best_run_latency_s is None + + def test_a_pending_trace_link_does_not_block_the_next_item(): """The overlap claim, proved by construction rather than by a stopwatch. diff --git a/packages/gooddata-eval/tests/test_agentic_search_tool.py b/packages/gooddata-eval/tests/test_agentic_search_tool.py index 975004cd4..f1100d9c9 100644 --- a/packages/gooddata-eval/tests/test_agentic_search_tool.py +++ b/packages/gooddata-eval/tests/test_agentic_search_tool.py @@ -208,6 +208,7 @@ def test_evaluate_agentic_search_tool_returns_reasoning_steps_on_pass(): "tool_correct": True, "tool_call_names": ["search_objects"], "latency_breakdown": [], + "tool_calls": [], } @@ -244,4 +245,5 @@ def test_evaluate_agentic_search_tool_attaches_reasoning_steps_to_exception_on_f "tool_correct": False, "tool_call_names": [], "latency_breakdown": [], + "tool_calls": [], } diff --git a/packages/gooddata-eval/tests/test_agentic_what_if.py b/packages/gooddata-eval/tests/test_agentic_what_if.py index 2f45aea4f..c0bd5e0f8 100644 --- a/packages/gooddata-eval/tests/test_agentic_what_if.py +++ b/packages/gooddata-eval/tests/test_agentic_what_if.py @@ -13,7 +13,7 @@ evaluate_agentic_what_if, run_agentic_what_if, ) -from gooddata_eval.core.models import ChatResult +from gooddata_eval.core.models import ChatResult, LoopExit _MODULE = "gooddata_eval.core.agentic.what_if" @@ -215,6 +215,7 @@ def test_a_single_turn_scenario_passes(): summary = _run([_chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is False + assert summary.best.exit_reason is LoopExit.SUCCESS def test_a_clarifying_question_is_answered_and_the_run_continues(): @@ -222,12 +223,15 @@ def test_a_clarifying_question_is_answered_and_the_run_continues(): summary = _run([_chat([], text="Which Spend metric did you mean?"), _chat(_pair())]) assert summary.pass_at_k is True assert summary.best.evaluation.disambiguated is True + assert summary.best.exit_reason is LoopExit.SUCCESS def test_the_loop_stops_at_max_iterations_without_a_scenario(): + # triggered=False alone cannot tell this apart from a refusal -- exit_reason can. summary = _run([_chat([], text="Still thinking")] * 4, max_iterations=4) assert summary.pass_at_k is False assert summary.best.evaluation.triggered is False + assert summary.best.exit_reason is LoopExit.BUDGET_EXHAUSTED def test_an_empty_response_ends_the_run_immediately(): @@ -238,15 +242,19 @@ def test_an_empty_response_ends_the_run_immediately(): patch(f"{_MODULE}.ChatClient", return_value=client), patch(f"{_MODULE}.generate_simulated_what_if_response") as sim, ): - run_agentic_what_if(host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED) + summary = run_agentic_what_if( + host="http://h", token="tok", workspace_id="ws1", question="q", expected_output=_EXPECTED + ) assert client.send_message.call_count == 1 sim.assert_not_called() + assert summary.best.exit_reason is LoopExit.AGENT_SILENT def test_a_chat_error_on_a_later_run_does_not_discard_the_earlier_one(): summary = _run([_chat(_pair()), RuntimeError("boom")], k=2) assert len(summary.run_results) == 2 assert summary.pass_at_k is True + assert summary.run_results[1].exit_reason is LoopExit.CHAT_ERROR def test_k_must_be_at_least_one(): @@ -277,6 +285,8 @@ def test_detail_reports_the_adjustments_the_agent_asked_for(): assert outcome.detail["actual_scenario_labels"] == ["Scenario A"] assert outcome.detail["asserted"] == ["metric_id", "scenario_maql"] assert "latency_breakdown" in outcome.detail + assert "tool_calls" in outcome.detail + assert outcome.detail["exit_reason"] == "success" assert outcome.runs_passed == 1 diff --git a/packages/gooddata-eval/tests/test_timing.py b/packages/gooddata-eval/tests/test_timing.py index 3b31a8411..de0951af8 100644 --- a/packages/gooddata-eval/tests/test_timing.py +++ b/packages/gooddata-eval/tests/test_timing.py @@ -9,6 +9,7 @@ TIMERS_ENV_VAR, PhaseTimings, log_timer, + run_latency_s, sum_timings, timers_enabled, ) @@ -91,3 +92,10 @@ def test_as_dict_rounds_every_phase(): def test_sum_timings_of_nothing_is_all_zeroes(): assert sum_timings([]) == PhaseTimings() + + +def test_run_latency_s_excludes_langfuse(): + # langfuse_s is deliberately off the critical path (see PhaseTimings docstring), so a + # trace-link slowdown must not inflate the run latency a worst-case table sorts on. + timings = PhaseTimings(agent_s=2.0, judge_s=1.0, simulated_user_s=0.5, langfuse_s=100.0) + assert run_latency_s(timings) == 3.5