From 389905b83f2b64267a9718cdc8f2327aa6f0bb50 Mon Sep 17 00:00:00 2001 From: Peter Tomko Date: Tue, 4 Aug 2026 14:00:57 +0200 Subject: [PATCH 1/3] feat(gooddata-eval): capture agent reasoning steps in ChatResult The SSE reasoning events were already being read to produce reasoning_step_count, but the step text itself was discarded. Keep it as reasoning_steps on ChatResult/ItemReport and surface it in the JSON report so eval consumers can inspect the agent's actual reasoning trace, not just how many steps it took. --- .../src/gooddata_eval/core/chat/sse_client.py | 1 + .../src/gooddata_eval/core/models.py | 1 + .../core/reporting/json_report.py | 1 + .../src/gooddata_eval/core/runner.py | 2 ++ .../gooddata-eval/tests/test_reporting.py | 3 +++ packages/gooddata-eval/tests/test_runner.py | 19 +++++++++++++++++++ .../gooddata-eval/tests/test_sse_client.py | 8 ++++++++ 7 files changed, 35 insertions(+) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py index 2db50d5a2..c6ce17e6c 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py @@ -171,6 +171,7 @@ def _build_chat_result(acc: _SseAccumulator) -> ChatResult: "alertProposals": acc.alert_proposals, "toolCallEvents": acc.tool_call_events, "reasoningStepCount": len(acc.reasoning_steps), + "reasoningSteps": [step["summary"] for step in acc.reasoning_steps], } if acc.visualizations: payload["createdVisualizations"] = { diff --git a/packages/gooddata-eval/src/gooddata_eval/core/models.py b/packages/gooddata-eval/src/gooddata_eval/core/models.py index 336c313b9..798685638 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/models.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/models.py @@ -98,6 +98,7 @@ class ChatResult(BaseModel): alert_proposals: list[dict] = Field(default_factory=list, alias="alertProposals") tool_call_events: list[ToolCallEvent] = Field(default_factory=list, alias="toolCallEvents") reasoning_step_count: int = Field(default=0, alias="reasoningStepCount") + reasoning_steps: list[str] = Field(default_factory=list, alias="reasoningSteps") conversation_id: str | None = Field(default=None, alias="conversationId") response_id: str | None = Field(default=None, alias="responseId") diff --git a/packages/gooddata-eval/src/gooddata_eval/core/reporting/json_report.py b/packages/gooddata-eval/src/gooddata_eval/core/reporting/json_report.py index d4c7b4a3e..1a28e0001 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/reporting/json_report.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/reporting/json_report.py @@ -36,6 +36,7 @@ def _build_run_dict(report: EvalReport) -> dict: "detail": item.best_detail, "conversation_id": item.conversation_id, "response_id": item.response_id, + "reasoning": item.reasoning_steps, } for item in report.items }, diff --git a/packages/gooddata-eval/src/gooddata_eval/core/runner.py b/packages/gooddata-eval/src/gooddata_eval/core/runner.py index 5bd56c9bf..0161cc3f8 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/runner.py @@ -33,6 +33,7 @@ class ItemReport: best_detail: dict = field(default_factory=dict) conversation_id: str | None = None response_id: str | None = None + reasoning_steps: list[str] = field(default_factory=list) @property def avg_latency_s(self) -> float: @@ -116,6 +117,7 @@ def _run_one_item( chat_result = backend.ask(item) report.conversation_id = getattr(chat_result, "conversation_id", None) or report.conversation_id report.response_id = getattr(chat_result, "response_id", None) or report.response_id + report.reasoning_steps = getattr(chat_result, "reasoning_steps", None) or report.reasoning_steps evaluation = evaluator.evaluate(item, chat_result) latency = time.perf_counter() - t0 report.runs += 1 diff --git a/packages/gooddata-eval/tests/test_reporting.py b/packages/gooddata-eval/tests/test_reporting.py index f191de38e..467fda7fb 100644 --- a/packages/gooddata-eval/tests/test_reporting.py +++ b/packages/gooddata-eval/tests/test_reporting.py @@ -23,6 +23,7 @@ def _report() -> EvalReport: pass_at_k=True, runs=2, latency_s=2.5, + reasoning_steps=["step one", "step two"], ), ItemReport( id="i2", @@ -48,6 +49,8 @@ def test_build_json_report_keyed_by_item_id(): assert data["items"]["i1"]["pass_at_k"] is True assert data["items"]["i1"]["latency_s"] == 2.5 assert data["items"]["i1"]["avg_latency_s"] == 1.25 + assert data["items"]["i1"]["reasoning"] == ["step one", "step two"] + assert data["items"]["i2"]["reasoning"] == [] def test_write_json_report_creates_file(tmp_path): diff --git a/packages/gooddata-eval/tests/test_runner.py b/packages/gooddata-eval/tests/test_runner.py index a3f1742b1..23ff8c64d 100644 --- a/packages/gooddata-eval/tests/test_runner.py +++ b/packages/gooddata-eval/tests/test_runner.py @@ -258,6 +258,25 @@ def ask(self, item: DatasetItem) -> ChatResult: assert "conversation_id" not in report.items[0].error +def test_run_items_carries_reasoning_steps_from_chat_result(): + """reasoning_steps from the ChatResult surfaces on the item report, same as conversation_id.""" + + class _ReasoningBackend: + def ask(self, item: DatasetItem) -> ChatResult: + return ChatResult.model_validate( + {"textResponse": "which metric?", "reasoningSteps": ["step one", "step two"]} + ) + + report = run_items([_item()], _ReasoningBackend(), runs=1) + assert report.items[0].reasoning_steps == ["step one", "step two"] + + +def test_run_items_reasoning_steps_empty_when_chat_result_has_none(): + backend = _FakeBackend([_empty_chat()]) + report = run_items([_item()], backend, runs=1) + assert report.items[0].reasoning_steps == [] + + def test_run_items_callback_exception_is_logged_not_swallowed(capsys): """A raising callback prints a traceback to stderr but the run continues.""" backend = _FakeBackend([_chat_with(_viz_obj())] * 2) diff --git a/packages/gooddata-eval/tests/test_sse_client.py b/packages/gooddata-eval/tests/test_sse_client.py index 490dfd57d..a8bb9419b 100644 --- a/packages/gooddata-eval/tests/test_sse_client.py +++ b/packages/gooddata-eval/tests/test_sse_client.py @@ -60,9 +60,17 @@ def test_parse_sse_lines_counts_reasoning_steps(): ] result = parse_sse_lines(lines) assert result.reasoning_step_count == 2 + assert result.reasoning_steps == ["step one", "step two"] assert result.text_response == "Done" +def test_parse_sse_lines_reasoning_steps_empty_when_no_reasoning_events(): + lines = ['data: {"item": {"role": "assistant", "content": {"type": "text", "text": "Done"}}}'] + result = parse_sse_lines(lines) + assert result.reasoning_step_count == 0 + assert result.reasoning_steps == [] + + def test_parse_sse_lines_prefers_multipart_viz_over_adhoc_fallback(): """Real multipart visualization takes priority over adhoc tool call stash.""" From da26fe4bcabc4b0ffedb02a31967b69347f0395b Mon Sep 17 00:00:00 2001 From: Peter Tomko Date: Tue, 4 Aug 2026 14:12:51 +0200 Subject: [PATCH 2/3] test(gooddata-eval): cover multi-run reasoning_steps retention Addresses CodeRabbit nitpick on PR #1708: a later run returning no reasoning events must not clobber an earlier run's captured steps (runner.py:120's `or` pattern, same as conversation_id/response_id). --- packages/gooddata-eval/tests/test_runner.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/packages/gooddata-eval/tests/test_runner.py b/packages/gooddata-eval/tests/test_runner.py index 23ff8c64d..925c214a6 100644 --- a/packages/gooddata-eval/tests/test_runner.py +++ b/packages/gooddata-eval/tests/test_runner.py @@ -277,6 +277,22 @@ def test_run_items_reasoning_steps_empty_when_chat_result_has_none(): assert report.items[0].reasoning_steps == [] +def test_run_items_reasoning_steps_keeps_earlier_run_when_later_run_is_empty(): + """A later run with no reasoning events must not clobber an earlier run's steps (runner.py:120's `or`).""" + + class _MixedReasoningBackend: + def __init__(self): + self.calls = 0 + + def ask(self, item: DatasetItem) -> ChatResult: + self.calls += 1 + steps = ["step one", "step two"] if self.calls == 1 else [] + return ChatResult.model_validate({"textResponse": "answer", "reasoningSteps": steps}) + + report = run_items([_item()], _MixedReasoningBackend(), runs=2) + assert report.items[0].reasoning_steps == ["step one", "step two"] + + def test_run_items_callback_exception_is_logged_not_swallowed(capsys): """A raising callback prints a traceback to stderr but the run continues.""" backend = _FakeBackend([_chat_with(_viz_obj())] * 2) From 8010bd44f00c508fa0a51e85da072a2eb646e577 Mon Sep 17 00:00:00 2001 From: Peter Tomko Date: Wed, 5 Aug 2026 09:13:35 +0200 Subject: [PATCH 3/3] feat(gooddata-eval): capture reasoning_steps through the agentic-CLI path 6001d2fd wired ChatResult.reasoning_steps through runner.py's generic single-turn path only. The agentic-CLI path (cli/agentic_runner.py -> evaluate_agentic_*) builds its own ItemReport and never touched it, so agentic_alert_skill/agentic_metric_skill/agentic_conversation items could never produce a reasoning trace, no matter what the platform emitted. Accumulates reasoning_steps across every send_message call in each of the three evaluators' run loops, attaches it to the run/turn result, and surfaces it from evaluate_agentic_* either as the return value (pass) or as an attribute on the raised exception (fail) -- mirroring the existing conversation_id-on-exception idiom in ChatClient.ask(). run_agentic_items picks it up from either path onto ItemReport.reasoning_steps, which json_report.py already serializes unconditionally. general_question/guardrail/search_tool/visualization are left untouched -- their evaluate_agentic_* functions still return None, unchanged. --- .../src/gooddata_eval/cli/agentic_runner.py | 28 ++-- .../gooddata_eval/core/agentic/alert_skill.py | 21 ++- .../core/agentic/conversation.py | 22 ++- .../core/agentic/metric_skill.py | 21 ++- .../tests/test_agentic_alert_skill.py | 115 ++++++++++++++++ .../tests/test_agentic_conversation.py | 126 ++++++++++++++++++ .../tests/test_agentic_metric_skill.py | 99 ++++++++++++++ .../tests/test_agentic_runner.py | 77 +++++++++++ 8 files changed, 487 insertions(+), 22 deletions(-) create mode 100644 packages/gooddata-eval/tests/test_agentic_runner.py diff --git a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py index f73b44679..6edf78fd1 100644 --- a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py @@ -83,8 +83,12 @@ def _dispatch_agentic( run_ts: str, model_version_override: str | None, reasoning_effort: ReasoningEffort | None = None, -) -> None: - """Call the appropriate evaluate_agentic_* function for the item's test_kind.""" +) -> list[str] | None: + """Call the appropriate evaluate_agentic_* function for the item's test_kind. + + Returns whatever that function returns -- only alert_skill/metric_skill/conversation + currently return their reasoning_steps; the rest still return None (unchanged). + """ kind = item.test_kind eo = item.expected_output lf_kw: _LfKw = { @@ -97,7 +101,7 @@ def _dispatch_agentic( } if kind in ("vis_agentic", "agentic_visualization"): - evaluate_agentic_visualization( + return evaluate_agentic_visualization( host=host, token=token, workspace_id=workspace_id, @@ -107,7 +111,7 @@ def _dispatch_agentic( **lf_kw, ) elif kind == "agentic_metric_skill": - evaluate_agentic_metric_skill( + return evaluate_agentic_metric_skill( host=host, token=token, workspace_id=workspace_id, @@ -117,7 +121,7 @@ def _dispatch_agentic( **lf_kw, ) elif kind == "agentic_alert_skill": - evaluate_agentic_alert_skill( + return evaluate_agentic_alert_skill( host=host, token=token, workspace_id=workspace_id, @@ -130,7 +134,7 @@ def _dispatch_agentic( eo_dict = eo if isinstance(eo, dict) else {} tool_call = eo_dict.get("tool_call", {}) expected_args = tool_call.get("function_arguments", eo_dict) - evaluate_agentic_search_tool( + return evaluate_agentic_search_tool( host=host, token=token, workspace_id=workspace_id, @@ -140,7 +144,7 @@ def _dispatch_agentic( **lf_kw, ) elif kind == "agentic_general_question": - evaluate_agentic_general_question( + return evaluate_agentic_general_question( host=host, token=token, workspace_id=workspace_id, @@ -150,7 +154,7 @@ def _dispatch_agentic( **lf_kw, ) elif kind == "agentic_guardrail": - evaluate_agentic_guardrail( + return evaluate_agentic_guardrail( host=host, token=token, workspace_id=workspace_id, @@ -161,7 +165,7 @@ def _dispatch_agentic( ) elif kind == "agentic_conversation": fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {} - evaluate_agentic_conversation( + return evaluate_agentic_conversation( host=host, token=token, workspace_id=workspace_id, @@ -207,12 +211,16 @@ def run_agentic_items( ) t0 = time.perf_counter() try: - _dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort) + reasoning_steps = _dispatch_agentic( + item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort + ) item_report.pass_at_k = True item_report.runs = k + item_report.reasoning_steps = reasoning_steps or [] except AssertionError as exc: item_report.pass_at_k = False item_report.runs = k + item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or [] print(f"[agentic] {item.id} FAIL: {exc}", flush=True) except Exception as exc: item_report.error = f"{type(exc).__name__}: {exc}" diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py index e75a2bed2..d82ad1600 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py @@ -6,7 +6,7 @@ import json import os import re -from dataclasses import dataclass +from dataclasses import dataclass, field from typing import Any from gooddata_sdk import GoodDataSdk @@ -233,6 +233,7 @@ class AlertRunResult: alert_id: str | None eval: AlertEvaluation actual_alert_arguments: dict + reasoning_steps: list[str] = field(default_factory=list) @dataclass @@ -357,6 +358,7 @@ def _run_once(conv_id: str) -> AlertRunResult: alert_id: str | None = None actual_args: dict = {} tool_called = False + reasoning_steps: list[str] = [] # conversation_history stores prior turns for GPT-4o context. # Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply. conversation_history: list = [] @@ -364,6 +366,7 @@ def _run_once(conv_id: str) -> AlertRunResult: for _iteration in range(max_iterations): chat_result = client.send_message(conv_id, current_question) + reasoning_steps.extend(chat_result.reasoning_steps or []) alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or []) if tool_called: alert_id_to_delete = alert_id @@ -397,6 +400,7 @@ def _run_once(conv_id: str) -> AlertRunResult: alert_id=alert_id, eval=ev, actual_alert_arguments=actual_args, + reasoning_steps=reasoning_steps, ) finally: if alert_id_to_delete: @@ -447,6 +451,7 @@ class AlertSkillAssertionError(AssertionError): """Raised when an alert-skill evaluation fails.""" __tracebackhide__ = True + reasoning_steps: list[str] def evaluate_agentic_alert_skill( @@ -465,8 +470,13 @@ def evaluate_agentic_alert_skill( model_version_override: str | None = None, run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, -) -> None: - """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.""" +) -> list[str]: + """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure. + + Returns the best run's reasoning_steps on success; on failure the same list is attached + to the raised exception as ``.reasoning_steps`` (mirrors the `conversation_id`-on-exception + idiom in `ChatClient.ask()`) so callers can retrieve it either way. + """ from datetime import datetime as _dt # noqa: PLC0415 from datetime import timezone as _tz # noqa: PLC0415 @@ -539,7 +549,7 @@ def evaluate_agentic_alert_skill( if not summary.pass_at_k: best = summary.best ev = best.eval - raise AlertSkillAssertionError( + exc = AlertSkillAssertionError( f"Alert skill assertion failed. strict_pass={ev.strict_pass}. " f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, " f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, " @@ -547,3 +557,6 @@ def evaluate_agentic_alert_skill( f"recipients_correct={ev.recipients_correct}. " f"Actual args: {best.actual_alert_arguments}" ) + exc.reasoning_steps = best.reasoning_steps + raise exc + return summary.best.reasoning_steps diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py index a7c3034fe..0715f3611 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py @@ -5,7 +5,7 @@ import json import re -from dataclasses import dataclass +from dataclasses import dataclass, field from typing import Literal from gooddata_sdk import GoodDataSdk @@ -270,6 +270,7 @@ class ConversationResult: full_skill_coverage: bool conversation_success: bool total_clarification_turns: int + reasoning_steps: list[str] = field(default_factory=list) def run_agentic_conversation( @@ -298,6 +299,7 @@ def run_agentic_conversation( # not persist in the (shared) workspace and get reused by a later test. Deferred to # the end — a later turn may $ref a metric an earlier turn created. created_metric_ids: list[str] = [] + reasoning_steps: list[str] = [] try: if initial_conversation_id is not None: @@ -320,6 +322,7 @@ def run_agentic_conversation( chat_result = client.send_message(conversation_id, current_message) final_result = chat_result all_tool_calls.extend(chat_result.tool_call_events or []) + reasoning_steps.extend(chat_result.reasoning_steps or []) if _check_output_present(resolved_turn, chat_result): break @@ -383,6 +386,7 @@ def run_agentic_conversation( full_skill_coverage=full_skill_coverage, conversation_success=conversation_success, total_clarification_turns=total_clarification_turns, + reasoning_steps=reasoning_steps, ) @@ -390,6 +394,7 @@ class ConversationAssertionError(AssertionError): """Raised when a conversation evaluation fails.""" __tracebackhide__ = True + reasoning_steps: list[str] def evaluate_agentic_conversation( @@ -406,8 +411,14 @@ def evaluate_agentic_conversation( model_version_override: str | None = None, run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, -) -> None: - """Run conversation evaluation, log to Langfuse, and raise on failure.""" +) -> list[str]: + """Run conversation evaluation, log to Langfuse, and raise on failure. + + Returns the conversation's reasoning_steps on success; on failure the same list is + attached to the raised exception as ``.reasoning_steps`` (mirrors the + `conversation_id`-on-exception idiom in `ChatClient.ask()`) so callers can retrieve it + either way. + """ from datetime import datetime as _dt # noqa: PLC0415 from datetime import timezone as _tz # noqa: PLC0415 @@ -483,8 +494,11 @@ def evaluate_agentic_conversation( if not result.conversation_success: failed_turns = [tr for tr in result.turn_results if not tr.skill_success] - raise ConversationAssertionError( + exc = ConversationAssertionError( f"Conversation assertion failed. " f"full_skill_coverage={result.full_skill_coverage}. " f"Failed turns: {[t.turn_id for t in failed_turns]}" ) + exc.reasoning_steps = result.reasoning_steps + raise exc + return result.reasoning_steps diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py index 2e2b5b9b1..553dced6d 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py @@ -5,7 +5,7 @@ import os import re -from dataclasses import dataclass +from dataclasses import dataclass, field from typing import Any from gooddata_sdk import GoodDataSdk @@ -109,6 +109,7 @@ class MetricRunResult: actual_maql: str maql_correct: bool total_turns: float + reasoning_steps: list[str] = field(default_factory=list) @dataclass @@ -192,11 +193,13 @@ def _execute_single_metric_run( metric_id_to_delete: str | None = None turns = 0 current_question = question + reasoning_steps: list[str] = [] try: for _iteration in range(max_iterations): turns += 1 chat_result = client.send_message(conversation_id, current_question) + reasoning_steps.extend(chat_result.reasoning_steps or []) candidate = _extract_metric_result(chat_result.tool_call_events or []) if candidate is not None: metric_result = candidate @@ -218,6 +221,7 @@ def _execute_single_metric_run( actual_maql=actual_maql, maql_correct=maql_correct, total_turns=float(turns), + reasoning_steps=reasoning_steps, ) finally: if metric_id_to_delete: @@ -285,6 +289,7 @@ class MetricSkillAssertionError(AssertionError): """Raised when a metric-skill evaluation fails.""" __tracebackhide__ = True + reasoning_steps: list[str] def evaluate_agentic_metric_skill( @@ -303,8 +308,13 @@ def evaluate_agentic_metric_skill( model_version_override: str | None = None, run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, -) -> None: - """Run metric-skill evaluation, log to Langfuse, and raise MetricSkillAssertionError on failure.""" +) -> list[str]: + """Run metric-skill evaluation, log to Langfuse, and raise MetricSkillAssertionError on failure. + + Returns the best run's reasoning_steps on success; on failure the same list is attached + to the raised exception as ``.reasoning_steps`` (mirrors the `conversation_id`-on-exception + idiom in `ChatClient.ask()`) so callers can retrieve it either way. + """ from datetime import datetime as _dt # noqa: PLC0415 from datetime import timezone as _tz # noqa: PLC0415 @@ -368,9 +378,12 @@ def evaluate_agentic_metric_skill( best = summary.best expected_outputs_list: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output] candidates_str = "; ".join(repr(c.get("maql", "")) for c in expected_outputs_list) - raise MetricSkillAssertionError( + exc = MetricSkillAssertionError( f"Metric skill assertion failed. " f"metric_created={best.metric_created}, maql_correct={best.maql_correct}. " f"Expected MAQL (candidates): {candidates_str}. " f"Actual MAQL: {best.actual_maql}." ) + exc.reasoning_steps = best.reasoning_steps + raise exc + return summary.best.reasoning_steps diff --git a/packages/gooddata-eval/tests/test_agentic_alert_skill.py b/packages/gooddata-eval/tests/test_agentic_alert_skill.py index fa5dcbedd..bb12bc2e5 100644 --- a/packages/gooddata-eval/tests/test_agentic_alert_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_alert_skill.py @@ -2,12 +2,15 @@ # SPDX-License-Identifier: LicenseRef-GoodData-Enterprise from unittest.mock import MagicMock, patch +import pytest from gooddata_eval.core.agentic.alert_skill import ( AlertEvaluation, + AlertSkillAssertionError, _check_trigger, _deep_subset, _normalize_expected_output, _to_number, + evaluate_agentic_alert_skill, render_alert_proposal, run_agentic_alert_skill, ) @@ -248,3 +251,115 @@ def test_run_agentic_alert_skill_answers_proposal_only_confirmation_turn(): assert "admin@gooddata.com" in agent_message assert summary.best.eval.alert_created is True assert summary.best.alert_id == "alert-1" + + +def test_run_agentic_alert_skill_accumulates_reasoning_steps_across_iterations(): + proposal_turn = ChatResult.model_validate( + { + "text_response": None, + "alertProposals": [_PROPOSAL], + "toolCallEvents": [ + {"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None} + ], + "reasoningSteps": ["step one"], + } + ) + created_turn = ChatResult.model_validate( + { + "text_response": "Alert created.", + "toolCallEvents": [ + { + "functionName": "create_metric_alert", + "functionArguments": '{"operator": "GREATER_THAN", "threshold": 500}', + "result": '{"id": "alert-1"}', + } + ], + "reasoningSteps": ["step two"], + } + ) + mock_client = MagicMock() + mock_client.send_message.side_effect = [proposal_turn, created_turn] + + with ( + patch("gooddata_eval.core.agentic.alert_skill.ChatClient", return_value=mock_client), + patch( + "gooddata_eval.core.agentic.alert_skill.generate_simulated_alert_response", + return_value="Yes, please proceed to create the alert.", + ), + patch("gooddata_eval.core.agentic.alert_skill._delete_alert"), + ): + summary = run_agentic_alert_skill( + host="http://host", + token="tok", + workspace_id="ws1", + question="Notify me whenever the number of orders goes above 500", + expected_output={"operator": "GREATER_THAN", "threshold": 500}, + k=1, + max_iterations=6, + initial_conversation_id="conv-1", + ) + + assert summary.best.reasoning_steps == ["step one", "step two"] + + +def test_evaluate_agentic_alert_skill_returns_reasoning_steps_on_pass(): + chat_result = ChatResult.model_validate( + { + "text_response": "Alert created.", + "toolCallEvents": [ + { + "functionName": "create_metric_alert", + "functionArguments": '{"operator": "GREATER_THAN", "threshold": 500}', + "result": '{"id": "alert-1"}', + } + ], + "reasoningSteps": ["thinking about it"], + } + ) + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + mock_client.send_message.return_value = chat_result + + with ( + patch("gooddata_eval.core.agentic.alert_skill.ChatClient", return_value=mock_client), + patch("gooddata_eval.core.agentic.alert_skill._delete_alert"), + ): + reasoning = evaluate_agentic_alert_skill( + host="http://host", + token="tok", + workspace_id="ws1", + question="Notify me whenever the number of orders goes above 500", + expected_output={"operator": "GREATER_THAN", "threshold": 500}, + k=1, + max_iterations=1, + ) + + assert reasoning == ["thinking about it"] + + +def test_evaluate_agentic_alert_skill_attaches_reasoning_steps_to_exception_on_fail(): + chat_result = ChatResult.model_validate( + { + "text_response": "I cannot create the alert", + "toolCallEvents": [], + "reasoningSteps": ["confused thinking"], + } + ) + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + mock_client.send_message.return_value = chat_result + + with ( + patch("gooddata_eval.core.agentic.alert_skill.ChatClient", return_value=mock_client), + pytest.raises(AlertSkillAssertionError) as exc_info, + ): + evaluate_agentic_alert_skill( + host="http://host", + token="tok", + workspace_id="ws1", + question="Create alert", + expected_output={"operator": "GREATER_THAN", "threshold": 100}, + k=1, + max_iterations=1, + ) + assert exc_info.value.reasoning_steps == ["confused thinking"] diff --git a/packages/gooddata-eval/tests/test_agentic_conversation.py b/packages/gooddata-eval/tests/test_agentic_conversation.py index 9d2234f33..5ea4eee8b 100644 --- a/packages/gooddata-eval/tests/test_agentic_conversation.py +++ b/packages/gooddata-eval/tests/test_agentic_conversation.py @@ -4,10 +4,12 @@ import pytest from gooddata_eval.core.agentic.conversation import ( + ConversationAssertionError, ConversationFixture, TurnDefinition, TurnResult, _resolve_refs, + evaluate_agentic_conversation, run_agentic_conversation, ) from gooddata_eval.core.models import ChatResult, ToolCallEvent @@ -352,3 +354,127 @@ def test_run_agentic_conversation_treats_alert_proposal_as_a_clarification(): assert "Should I create this alert?" in mock_sim.call_args.args[0] assert result.turn_results[0].clarification_turns_used == 1 assert result.turn_results[0].skill_success is True + + +def test_run_agentic_conversation_accumulates_reasoning_steps_across_turns(): + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + tc = MagicMock(spec=ToolCallEvent) + tc.function_name = "set_skills" + tc.parsed_arguments = lambda: {"skills": ["visualization"]} + + turn1_result = MagicMock() + turn1_result.text_response = "Here is your visualization" + turn1_result.created_visualizations = [MagicMock()] + turn1_result.tool_call_events = [tc] + turn1_result.reasoning_steps = ["turn one reasoning"] + + turn2_result = MagicMock() + turn2_result.text_response = "Here is another visualization" + turn2_result.created_visualizations = [MagicMock()] + turn2_result.tool_call_events = [tc] + turn2_result.reasoning_steps = ["turn two reasoning"] + + mock_client.send_message.side_effect = [turn1_result, turn2_result] + + fixture = ConversationFixture( + id="test-reasoning", + expected_skills=["visualization"], + turns=[ + TurnDefinition( + turn_id="t1", + message="Make a chart", + expected_skill="visualization", + expected_output_type="visualization", + ), + TurnDefinition( + turn_id="t2", + message="Make another chart", + expected_skill="visualization", + expected_output_type="visualization", + ), + ], + ) + with patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client): + result = run_agentic_conversation( + host="http://host/api/v1/actions/workspaces/ws1/ai", + token="tok", + workspace_id="ws1", + fixture=fixture, + ) + + assert result.reasoning_steps == ["turn one reasoning", "turn two reasoning"] + + +def test_evaluate_agentic_conversation_returns_reasoning_steps_on_pass(): + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + tc = MagicMock(spec=ToolCallEvent) + tc.function_name = "set_skills" + tc.parsed_arguments = lambda: {"skills": ["visualization"]} + chat_result = MagicMock() + chat_result.text_response = "Here is your visualization" + chat_result.created_visualizations = [MagicMock()] + chat_result.tool_call_events = [tc] + chat_result.reasoning_steps = ["thinking about it"] + mock_client.send_message.return_value = chat_result + + fixture = ConversationFixture( + id="test-1", + expected_skills=["visualization"], + turns=[ + TurnDefinition( + turn_id="t1", + message="Make a chart", + expected_skill="visualization", + expected_output_type="visualization", + ) + ], + ) + with patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client): + reasoning = evaluate_agentic_conversation( + host="http://host", + token="tok", + workspace_id="ws1", + fixture=fixture, + ) + assert reasoning == ["thinking about it"] + + +def test_evaluate_agentic_conversation_attaches_reasoning_steps_to_exception_on_fail(): + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + tc = MagicMock(spec=ToolCallEvent) + tc.function_name = "set_skills" + tc.parsed_arguments = lambda: {"skills": ["other_skill"]} + chat_result = MagicMock() + chat_result.text_response = "Here is something else" + chat_result.created_visualizations = None + chat_result.tool_call_events = [tc] + chat_result.alert_proposals = [] + chat_result.reasoning_steps = ["confused thinking"] + mock_client.send_message.return_value = chat_result + + fixture = ConversationFixture( + id="test-1", + expected_skills=["visualization"], + turns=[ + TurnDefinition( + turn_id="t1", + message="Make a chart", + expected_skill="visualization", + expected_output_type="visualization", + ) + ], + ) + with ( + patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client), + pytest.raises(ConversationAssertionError) as exc_info, + ): + evaluate_agentic_conversation( + host="http://host", + token="tok", + workspace_id="ws1", + fixture=fixture, + ) + assert exc_info.value.reasoning_steps == ["confused thinking"] diff --git a/packages/gooddata-eval/tests/test_agentic_metric_skill.py b/packages/gooddata-eval/tests/test_agentic_metric_skill.py index 67a163e92..19076dacf 100644 --- a/packages/gooddata-eval/tests/test_agentic_metric_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_metric_skill.py @@ -6,8 +6,10 @@ from gooddata_eval.core.agentic.metric_skill import ( AgenticMetricSummary, MetricRunResult, + MetricSkillAssertionError, _delete_metric, _normalize_maql, + evaluate_agentic_metric_skill, run_agentic_metric_skill, ) from gooddata_eval.core.models import ChatResult @@ -224,3 +226,100 @@ def test_run_agentic_metric_skill_deletes_metric_even_when_teardown_fails(): ) mock_sdk._client.entities_api.delete_entity_metrics.assert_called_once_with("ws1", "foo_metric") + + +def test_run_agentic_metric_skill_accumulates_reasoning_steps_across_iterations(): + clarify_turn = ChatResult.model_validate( + { + "textResponse": "Could you clarify which foo you mean?", + "toolCallEvents": [], + "reasoningSteps": ["step one"], + } + ) + created_turn = ChatResult.model_validate( + { + "textResponse": "done", + "toolCallEvents": [ + { + "functionName": "create_metric", + "functionArguments": "{}", + "result": '{"data": {"maql": "SELECT {metric/foo}"}}', + } + ], + "reasoningSteps": ["step two"], + } + ) + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + mock_client.send_message.side_effect = [clarify_turn, created_turn] + + with ( + patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client), + patch("gooddata_eval.core.agentic.metric_skill.generate_simulated_response", return_value="It's foo"), + ): + summary = run_agentic_metric_skill( + host="http://host/api/v1/actions/workspaces/ws1/ai", + token="tok", + workspace_id="ws1", + question="Create metric foo", + expected_output={"maql": "SELECT {metric/foo}"}, + k=1, + max_iterations=2, + ) + + assert summary.best.reasoning_steps == ["step one", "step two"] + + +def test_evaluate_agentic_metric_skill_returns_reasoning_steps_on_pass(): + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + mock_client.send_message.return_value = ChatResult.model_validate( + { + "textResponse": "done", + "toolCallEvents": [ + { + "functionName": "create_metric", + "functionArguments": "{}", + "result": '{"data": {"maql": "SELECT {metric/foo}"}}', + } + ], + "reasoningSteps": ["thinking about it"], + } + ) + with patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client): + reasoning = evaluate_agentic_metric_skill( + host="http://host/api/v1/actions/workspaces/ws1/ai", + token="tok", + workspace_id="ws1", + question="Create metric foo", + expected_output={"maql": "SELECT {metric/foo}"}, + k=1, + max_iterations=1, + ) + assert reasoning == ["thinking about it"] + + +def test_evaluate_agentic_metric_skill_attaches_reasoning_steps_to_exception_on_fail(): + mock_client = MagicMock() + mock_client.create_conversation.return_value = "conv-1" + mock_client.send_message.return_value = ChatResult.model_validate( + { + "textResponse": "I will work on that.", + "toolCallEvents": [], + "reasoningSteps": ["confused thinking"], + } + ) + with ( + patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client), + pytest.raises(MetricSkillAssertionError) as exc_info, + ): + evaluate_agentic_metric_skill( + host="http://host/api/v1/actions/workspaces/ws1/ai", + token="tok", + workspace_id="ws1", + question="Create metric foo", + expected_output={"maql": "SELECT {metric/foo}"}, + k=1, + max_iterations=1, + ) + assert exc_info.value.reasoning_steps == ["confused thinking"] diff --git a/packages/gooddata-eval/tests/test_agentic_runner.py b/packages/gooddata-eval/tests/test_agentic_runner.py new file mode 100644 index 000000000..9d6c5c70c --- /dev/null +++ b/packages/gooddata-eval/tests/test_agentic_runner.py @@ -0,0 +1,77 @@ +# (C) 2026 GoodData Corporation. All rights reserved. +# SPDX-License-Identifier: LicenseRef-GoodData-Enterprise +from unittest.mock import patch + +from gooddata_eval.cli.agentic_runner import run_agentic_items +from gooddata_eval.core.agentic.alert_skill import AlertSkillAssertionError +from gooddata_eval.core.models import DatasetItem + + +def _item(test_kind: str = "agentic_alert_skill") -> DatasetItem: + return DatasetItem( + id="item-1", + dataset_name="d", + test_kind=test_kind, + question="Alert me when revenue drops below 100.", + expected_output={"operator": "LESS_THAN", "threshold": 100}, + ) + + +def test_run_agentic_items_surfaces_reasoning_steps_on_pass(): + with patch( + "gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", + return_value=["it created the alert"], + ): + report = run_agentic_items( + [_item()], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + assert report.items[0].pass_at_k is True + assert report.items[0].reasoning_steps == ["it created the alert"] + + +def test_run_agentic_items_surfaces_reasoning_steps_from_exception_on_fail(): + exc = AlertSkillAssertionError("nope") + exc.reasoning_steps = ["it got confused"] + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", side_effect=exc): + report = run_agentic_items( + [_item()], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + assert report.items[0].pass_at_k is False + assert report.items[0].reasoning_steps == ["it got confused"] + + +def test_run_agentic_items_defaults_reasoning_steps_to_empty_when_exception_has_none(): + with patch( + "gooddata_eval.cli.agentic_runner.evaluate_agentic_alert_skill", + side_effect=AlertSkillAssertionError("nope"), + ): + report = run_agentic_items( + [_item()], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + assert report.items[0].reasoning_steps == [] + + +def test_run_agentic_items_defaults_reasoning_steps_to_empty_for_untouched_kinds(): + # general_question/guardrail/search_tool/visualization still return None -- unchanged. + with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_guardrail", return_value=None): + report = run_agentic_items( + [_item(test_kind="agentic_guardrail")], + host="http://host", + token="tok", + workspace_id="ws1", + run_ts="2026-01-01", + ) + assert report.items[0].pass_at_k is True + assert report.items[0].reasoning_steps == []