diff --git a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py index 20885a85c..370fc8f83 100644 --- a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py @@ -18,6 +18,7 @@ from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill +from gooddata_eval.core.agentic.report_skill import evaluate_agentic_report_skill from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization from gooddata_eval.core.agentic.what_if import evaluate_agentic_what_if @@ -43,6 +44,7 @@ class _LfKw(TypedDict, total=False): "agentic_metric_skill", "agentic_alert_skill", "agentic_dashboard_skill", + "agentic_report_skill", "agentic_search", "agentic_general_question", "agentic_guardrail", @@ -92,7 +94,8 @@ class _LfKw(TypedDict, total=False): # # agentic_dashboard_skill is absent by default rather than by evidence: gen-ai holds the draft and # any chart it authors in conversation state and writes neither until a user saves from the UI, so -# it is a candidate for the allowlist once the dataset has runs behind it. +# it is a candidate for the allowlist once the dataset has runs behind it. agentic_report_skill is +# absent for the same reason: the report draft stays in conversation state until a user saves it. WORKSPACE_MUTATING_TEST_KINDS = frozenset(AGENTIC_TEST_KINDS) - PARALLEL_SAFE_TEST_KINDS @@ -203,6 +206,18 @@ def _dispatch_agentic( agent_id=agent_id, **lf_kw, ) + elif kind == "agentic_report_skill": + return evaluate_agentic_report_skill( + host=host, + token=token, + workspace_id=workspace_id, + question=item.question, + expected_output=eo if isinstance(eo, dict) else {}, + k=k, + gate=gate, + agent_id=agent_id, + **lf_kw, + ) elif kind == "agentic_alert_skill": return evaluate_agentic_alert_skill( host=host, diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py index 3d74a4b9f..9ddbd1b20 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py @@ -53,6 +53,14 @@ evaluate_agentic_metric_skill, run_agentic_metric_skill, ) +from gooddata_eval.core.agentic.report_skill import ( + AgenticReportSummary, + ReportEvaluation, + ReportRunResult, + ReportSkillAssertionError, + evaluate_agentic_report_skill, + run_agentic_report_skill, +) from gooddata_eval.core.agentic.search_tool import ( AgenticSearchSummary, SearchResult, @@ -75,6 +83,7 @@ "AgenticGuardrailSummary", "AgenticKdaSummary", "AgenticMetricSummary", + "AgenticReportSummary", "AgenticSearchSummary", "AgenticRunSummary", "AlertEvaluation", @@ -95,6 +104,9 @@ "KdaSkillAssertionError", "MetricRunResult", "MetricSkillAssertionError", + "ReportEvaluation", + "ReportRunResult", + "ReportSkillAssertionError", "RunResult", "SearchResult", "SearchToolAssertionError", @@ -108,6 +120,7 @@ "evaluate_agentic_guardrail", "evaluate_agentic_kda_skill", "evaluate_agentic_metric_skill", + "evaluate_agentic_report_skill", "evaluate_agentic_search_tool", "evaluate_agentic_visualization", "run_agentic_alert_skill", @@ -117,6 +130,7 @@ "run_agentic_guardrail", "run_agentic_kda_skill", "run_agentic_metric_skill", + "run_agentic_report_skill", "run_agentic_search_tool", "run_agentic_visualization", ] diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py new file mode 100644 index 000000000..8c80b08d9 --- /dev/null +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py @@ -0,0 +1,639 @@ +# (C) 2026 GoodData Corporation. All rights reserved. +"""Agentic report-skill evaluation runner.""" + +from __future__ import annotations + +import time +from dataclasses import dataclass, field +from typing import Any + +from gooddata_eval.core.agentic._conversation_context import classify_reply +from gooddata_eval.core.agentic._gate import ( + DEFAULT_GATE, + EvalGate, + gate_failure_note, + gate_passed, + log_gate_scores, + stamp_gate_metadata, +) +from gooddata_eval.core.agentic._trace_linker import ( + RunIdentity, + RunTraceContext, + SubmitTraceLink, + open_trace_window, + run_trace_link_inline, + submit_trace_scoring, + utc_now, +) +from gooddata_eval.core.agentic.dashboard_skill import _extract_tool_result, _skill_activated +from gooddata_eval.core.chat.render import render_answer_text +from gooddata_eval.core.chat.sse_client import ChatClient +from gooddata_eval.core.config import ReasoningEffort +from gooddata_eval.core.models import ( + AgenticAssertionError, + AgenticEvalOutcome, + ChatResult, + ReasoningStepEvent, + ToolCallEvent, + build_latency_breakdown, + shift_and_index_events, +) +from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings + +_DEFAULT_K = 1 +# Same slack as the dashboard skill: the simulated reply is fixed, so the extra rounds only +# absorb a turn that answered without drafting. +_DEFAULT_MAX_ITERATIONS = 4 + +_DRAFT_TOOL = "draft_report" +_BUILDER_SKILL = "report_builder" +_PART_TYPE = "report" +# The copilot's own rule: a report is a cover plus at least one content page. +_MIN_PAGES = 2 + + +def _extract_report_part(chat_result: ChatResult) -> dict | None: + """The last ``report`` part of the turn, read back from ``unhandled_parts`` by type. + + The last one, because a turn that drafts and then refines leaves the refined version as + the one the user sees. + """ + for part in reversed(chat_result.unhandled_parts): + if isinstance(part, dict) and part.get("type") == _PART_TYPE: + return part + return None + + +def _visualizations_of(report: dict) -> set[str]: + """Every visualization id placed anywhere in the report's page layouts. + + A layout nests ``column`` and ``row`` lists to any depth before it reaches a slot, so the + walk goes through every dict and list rather than assuming a fixed shape. + """ + found: set[str] = set() + stack: list[Any] = list(report.get("pages") or []) + while stack: + node = stack.pop() + if isinstance(node, dict): + viz = node.get("visualization") + if isinstance(viz, str): + found.add(viz) + stack.extend(node.values()) + elif isinstance(node, list): + stack.extend(node) + return found + + +def _has_period(expected_output: dict) -> bool: + return "period" in expected_output + + +def _has_visualizations(expected_output: dict) -> bool: + return "visualizations" in expected_output + + +def _expects_clarification(expected_output: dict) -> bool: + return "expects_clarification" in expected_output + + +def _validate_expectation(expected_output: dict) -> None: + """Reject a fixture the run could not score meaningfully, before the first API call. + + Every key is optional. A key that is present has to be usable, because a malformed one + would otherwise score vacuously or only surface on the branch where the copilot asks back. + + Raises: + ValueError: the expectation is unusable. + """ + if _has_period(expected_output): + period = expected_output.get("period") + if not isinstance(period, dict) or not period.get("start") or not period.get("end"): + raise ValueError(f"period needs both 'start' and 'end', got {period!r}") + if _has_visualizations(expected_output): + visualizations = expected_output.get("visualizations") + if not isinstance(visualizations, list) or not visualizations: + raise ValueError("visualizations is empty; the chart check would pass vacuously") + for entry in visualizations: + if not isinstance(entry, dict) or not entry.get("id"): + raise ValueError(f"every visualization needs an 'id', got {entry!r}") + if _expects_clarification(expected_output) and not isinstance(expected_output["expects_clarification"], bool): + raise ValueError( + f"expects_clarification must be true or false, got {expected_output['expects_clarification']!r}" + ) + + +def build_simulated_reply(expected_output: dict) -> str: + """The reply the simulated user sends when the copilot asks back instead of drafting. + + Deterministic and LLM-free, built from what the fixture states, so a failure stays the + copilot's rather than a simulated user that phrased things differently each run. + """ + segments: list[str] = [] + visualizations = expected_output.get("visualizations") or [] + if visualizations: + titles = ", ".join(str(v.get("title") or v.get("id")) for v in visualizations) + segments.append(f"Please use these charts: {titles}.") + period = expected_output.get("period") + if isinstance(period, dict): + segments.append(f"Period: {period.get('start')} to {period.get('end')}.") + segments.append("Anything else is up to you. Please create the report now.") + return " ".join(segments) + + +@dataclass(frozen=True) +class _Applies: + """Which conditional checks the case applies; published only when they do. + + A check that could not fail is not evidence, and publishing it as passed would lift + ``quality_score`` above what the run earned. + """ + + period: bool + charts: bool + + +@dataclass +class ReportEvaluation: + """Per-run outcome of the report-skill checks; ``strict_checks`` is what the run is scored on.""" + + drafted: bool + part_present: bool + ref_matches: bool + pages_consistent: bool + not_saved: bool + skill_activated: bool + applies: _Applies + period_correct: bool = False + charts_matched: bool = False + failures: list[str] = field(default_factory=list) + + @property + def strict_pass(self) -> bool: + return all(self.strict_checks.values()) + + @property + def strict_checks(self) -> dict[str, bool]: + # Every name carries the `report_` prefix: the combo report resolves a trace's skill by + # which score names it carries, so a name shared with another skill would misfile it. + checks = { + "report_drafted": self.drafted, + "report_part_present": self.part_present, + "report_ref_matches": self.ref_matches, + "report_pages_consistent": self.pages_consistent, + "report_not_saved": self.not_saved, + "report_skill_activated": self.skill_activated, + } + if self.applies.period: + checks["report_period_correct"] = self.period_correct + if self.applies.charts: + checks["report_charts_matched"] = self.charts_matched + return checks + + +def _read_report(report_part: dict | None) -> tuple[dict | None, str | None]: + """The part's report document, or why it carries no usable one.""" + if report_part is None: + return None, f"the response carries no {_PART_TYPE!r} part" + report = report_part.get("report") + if not isinstance(report, dict): + return ( + None, + f"the {_PART_TYPE!r} part carries no report document (report_ref {report_part.get('report_ref')!r})", + ) + if report.get("type") != _PART_TYPE: + return None, ( + f"the {_PART_TYPE!r} part carries a document of type {report.get('type')!r}, expected {_PART_TYPE!r}" + ) + return report, None + + +def evaluate_report_response( + tool_result: dict | None, + report_part: dict | None, + expected_output: dict, + *, + skill_activated: bool, +) -> ReportEvaluation: + """Score one report response against its expectation. + + Pure: no network and no conversation state, so the whole assertion surface is unit-testable + without an agent. + """ + applies = _Applies(period=_has_period(expected_output), charts=_has_visualizations(expected_output)) + + if tool_result is None: + return ReportEvaluation( + drafted=False, + part_present=False, + ref_matches=False, + pages_consistent=False, + not_saved=False, + skill_activated=skill_activated, + applies=applies, + failures=[f"the agent never produced a successful {_DRAFT_TOOL} call"], + ) + + report, part_failure = _read_report(report_part) + if report is None or report_part is None: + return ReportEvaluation( + drafted=True, + part_present=False, + ref_matches=False, + pages_consistent=False, + not_saved=False, + skill_activated=skill_activated, + applies=applies, + failures=[part_failure] if part_failure else [], + ) + + failures: list[str] = [] + + part_ref, tool_ref = report_part.get("report_ref"), tool_result.get("ref") + ref_matches = isinstance(part_ref, str) and bool(part_ref) and part_ref == tool_ref + if not ref_matches: + failures.append(f"the {_PART_TYPE!r} part shows {part_ref!r}, but {_DRAFT_TOOL} returned {tool_ref!r}") + + pages = report.get("pages") or [] + part_count, tool_count = report_part.get("page_count"), tool_result.get("page_count") + if part_count != len(pages) or tool_count != len(pages): + failures.append( + f"the report has {len(pages)} page(s), but the part says {part_count} and {_DRAFT_TOOL} said {tool_count}" + ) + pages_consistent = False + elif len(pages) < _MIN_PAGES: + failures.append(f"the report has {len(pages)} page(s); it needs a cover and at least one content page") + pages_consistent = False + else: + pages_consistent = True + + not_saved = True + for key, wording in (("saved_report_id", "be saved yet"), ("base_report_id", "edit a saved report")): + value = report_part.get(key) + if value is not None: + failures.append(f"a new draft must not {wording}, but it reports {key} {value!r}") + not_saved = False + + period_correct = False + if applies.period: + expected_period = expected_output["period"] + actual_period = report.get("period") or {} + period_correct = all(actual_period.get(key) == expected_period.get(key) for key in ("start", "end")) + if not period_correct: + failures.append( + f"the report covers {actual_period.get('start')} to {actual_period.get('end')}, " + f"expected {expected_period.get('start')} to {expected_period.get('end')}" + ) + + charts_matched = False + if applies.charts: + placed = _visualizations_of(report) + missing = [v for v in expected_output["visualizations"] if v.get("id") not in placed] + failures.extend(f"the report does not show chart {v.get('id')!r} ({v.get('title')!r})" for v in missing) + charts_matched = not missing + + return ReportEvaluation( + drafted=True, + part_present=True, + ref_matches=ref_matches, + pages_consistent=pages_consistent, + not_saved=not_saved, + skill_activated=skill_activated, + applies=applies, + period_correct=period_correct, + charts_matched=charts_matched, + failures=failures, + ) + + +@dataclass +class ReportRunResult: + """Outcome of one conversation.""" + + conversation_id: str + evaluation: ReportEvaluation + expects_clarification: bool = False + asked_first: bool = False + tool_result: dict | None = None + report_part: dict | None = None + total_turns: int = 0 + total_steps: int = 0 + reasoning_steps: list[str] = field(default_factory=list) + response_id: str | None = None + tool_call_events: list[ToolCallEvent] = field(default_factory=list) + reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + timings: PhaseTimings = field(default_factory=PhaseTimings) + + @property + def diagnostics(self) -> dict[str, bool]: + """Observed but not scored. + + Whether the copilot asked before drafting is recorded only when the fixture says it + expects a question, and never gates: how much the copilot should ask is a product + decision, not something a run can get wrong. + """ + return {"report_asked_first": self.asked_first} if self.expects_clarification else {} + + @property + def summaries_from_data(self) -> int | None: + value = (self.tool_result or {}).get("summaries_from_data") + return value if isinstance(value, int) else None + + +@dataclass +class AgenticReportSummary: + """Aggregated outcome of K runs.""" + + run_results: list[ReportRunResult] + pass_at_k: bool + pass_power_k: bool + best: ReportRunResult + + +def _execute_single_report_run( + client: ChatClient, + conversation_id: str, + question: str, + expected_output: dict, + max_iterations: int, +) -> ReportRunResult: + """Drive one conversation until the copilot drafts a report, then evaluate it. + + Nothing is cleaned up on the way out by design: gen-ai keeps the draft in conversation + state and persists nothing until a user saves it. + """ + tool_result: dict | None = None + report_part: dict | None = None + turns = 0 + steps = 0 + current_question = question + reasoning_steps: list[str] = [] + response_id: str | None = None + all_tool_call_events: list[ToolCallEvent] = [] + all_reasoning_step_events: list[ReasoningStepEvent] = [] + timings = PhaseTimings() + turn_offset = 0.0 + tool_index_offset = 0 + reasoning_index_offset = 0 + first_turn_asked = False + + for iteration in range(max_iterations): + turns += 1 + agent_started = time.monotonic() + chat_result = client.send_message(conversation_id, current_question) + agent_elapsed = time.monotonic() - agent_started + timings.agent_s += agent_elapsed + reasoning_steps.extend(chat_result.reasoning_steps or []) + response_id = chat_result.response_id or response_id + turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events( + chat_result, + turn_offset=turn_offset, + tool_index_offset=tool_index_offset, + reasoning_index_offset=reasoning_index_offset, + ) + all_tool_call_events.extend(chat_result.tool_call_events or []) + all_reasoning_step_events.extend(chat_result.reasoning_step_events or []) + steps += chat_result.reasoning_step_count + + candidate = _extract_tool_result(chat_result.tool_call_events or [], _DRAFT_TOOL) + if candidate is not None: + log_timer( + f"[timer] report_skill {conversation_id} GoodData turn {turns} complete after " + f"{agent_elapsed:.2f}s; {_DRAFT_TOOL} result received" + ) + tool_result = candidate + # Read from the same turn as the draft: the part shows the version that call stored. + report_part = _extract_report_part(chat_result) + break + + response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) + if iteration == 0: + first_turn_asked = classify_reply(chat_result, response_text) == "question" + if not response_text and not chat_result.tool_call_events: + break + if iteration >= max_iterations - 1: + break + log_timer( + f"[timer] report_skill {conversation_id} GoodData turn {turns} complete after " + f"{agent_elapsed:.2f}s; answering with the expected charts and period" + ) + current_question = build_simulated_reply(expected_output) + + return ReportRunResult( + conversation_id=conversation_id, + evaluation=evaluate_report_response( + tool_result, + report_part, + expected_output, + skill_activated=_skill_activated(all_tool_call_events, _BUILDER_SKILL), + ), + expects_clarification=bool(expected_output.get("expects_clarification")), + asked_first=tool_result is not None and first_turn_asked, + tool_result=tool_result, + report_part=report_part, + total_turns=turns, + total_steps=steps, + reasoning_steps=reasoning_steps, + response_id=response_id, + tool_call_events=all_tool_call_events, + reasoning_step_events=all_reasoning_step_events, + timings=timings, + ) + + +def run_agentic_report_skill( + host: str, + token: str, + workspace_id: str, + question: str, + expected_output: dict, + k: int = _DEFAULT_K, + max_iterations: int = _DEFAULT_MAX_ITERATIONS, + initial_conversation_id: str | None = None, + reasoning_effort: ReasoningEffort | None = None, + agent_id: str | None = None, +) -> AgenticReportSummary: + """Run the report-skill agentic evaluation K times and return a summary. + + Raises: + ValueError: the fixture is unusable — see ``_validate_expectation``. + """ + _validate_expectation(expected_output) + run_results: list[ReportRunResult] = [] + client = ChatClient( + host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + ) + + try: + conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() + try: + run_results.append(_execute_single_report_run(client, conv_id_0, question, expected_output, max_iterations)) + finally: + if initial_conversation_id is None: # only delete conversations we created + client.delete_conversation(conv_id_0) + + for _ in range(1, k): + conv_id = client.create_conversation() + try: + run_results.append( + _execute_single_report_run(client, conv_id, question, expected_output, max_iterations) + ) + finally: + client.delete_conversation(conv_id) + finally: + client.close() + + return AgenticReportSummary( + run_results=run_results, + pass_at_k=any(r.evaluation.strict_pass for r in run_results), + pass_power_k=all(r.evaluation.strict_pass for r in run_results), + best=max(run_results, key=lambda r: sum(r.evaluation.strict_checks.values())), + ) + + +class ReportSkillAssertionError(AgenticAssertionError): + """Raised when a report-skill evaluation fails.""" + + +def evaluate_agentic_report_skill( + host: str, + token: str, + workspace_id: str, + question: str, + expected_output: dict, + k: int = _DEFAULT_K, + max_iterations: int = _DEFAULT_MAX_ITERATIONS, + initial_conversation_id: str | None = None, + agent_id: str | None = None, + langfuse: object | None = None, + dataset_item_id: str = "", + dataset_name: str = "report_skill", + run_timestamp: str | None = None, + model_version_override: str | None = None, + run_metadata_extra: dict | None = None, + reasoning_effort: ReasoningEffort | None = None, + submit_trace_link: SubmitTraceLink = run_trace_link_inline, + gate: EvalGate = DEFAULT_GATE, +) -> AgenticEvalOutcome: + """Run report-skill evaluation, log to Langfuse, and raise on failure. + + Returns the best run's outcome on success; on failure the same values are attached to the + raised ``ReportSkillAssertionError`` so callers can retrieve them either way. + + Raises: + ReportSkillAssertionError: the gate did not pass. + ValueError: the fixture is unusable — see ``_validate_expectation``. Raised before any + request, so it means a fixture to fix rather than a result to read. + """ + langfuse, window_start = open_trace_window(langfuse) + summary = run_agentic_report_skill( + host=host, + token=token, + workspace_id=workspace_id, + question=question, + expected_output=expected_output, + k=k, + max_iterations=max_iterations, + initial_conversation_id=initial_conversation_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + ) + + if langfuse is not None and dataset_item_id: + # Pinned on the calling thread: a deferred poll must not widen its query window. + window_end = utc_now() + + def _write_scores(ctx: RunTraceContext) -> None: + stamp_gate_metadata(ctx.run_metadata, k=len(summary.run_results), gate=gate) + + for run_idx, run in enumerate(summary.run_results): + pt = ctx.trace(run.conversation_id) + strict_checks = run.evaluation.strict_checks + with ctx.observe(pt, run_idx, conversation_id=run.conversation_id, output=strict_checks) as tid: + for score_name, value in strict_checks.items(): + ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN") + for name, value in run.diagnostics.items(): + ctx.score(tid, name=name, value=float(value), data_type="BOOLEAN") + if run.summaries_from_data is not None: + ctx.score( + tid, name="report_summaries_from_data", value=run.summaries_from_data, data_type="NUMERIC" + ) + ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC") + ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC") + log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k) + ctx.quality( + tid, + strict_checks=strict_checks, + latency_sec=pt.latency if pt else None, + cost_usd=pt.total_cost if pt else None, + ) + + # Before the pass@K raise: a failing item's scores are the ones worth having. + submit_trace_scoring( + submit_trace_link, + RunIdentity( + host, + token, + workspace_id, + dataset_name, + run_timestamp, + model_version_override, + run_metadata_extra, + reasoning_effort, + ), + langfuse=langfuse, + dataset_item_id=dataset_item_id, + conversation_ids=[r.conversation_id for r in summary.run_results], + window_start=window_start, + window_end=window_end, + suffix_runs=len(summary.run_results) > 1, + write_scores=_write_scores, + item_input=question, + ) + + item_timings = sum_timings([r.timings for r in summary.run_results]) + runs_passed = sum(1 for r in summary.run_results if r.evaluation.strict_pass) + runs_effective = len(summary.run_results) + + best = summary.best + detail: dict[str, Any] = { + **best.evaluation.strict_checks, + **best.diagnostics, + "summaries_from_data": best.summaries_from_data, + "turns": best.total_turns, + "failures": best.evaluation.failures, + "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), + } + + if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): + gate_note = gate_failure_note(gate, runs_passed, runs_effective) + skill_note = ( + "" + if best.evaluation.skill_activated + else ( + f" set_skills never activated {_BUILDER_SKILL}: either the copilot routed elsewhere, or the skill" + " is not registered. It registers only with the dashboard-builder deployment flag and the org's" + " enableBusinessBriefingReportsApp both on." + ) + ) + exc = ReportSkillAssertionError( + f"Report skill assertion failed. {gate_note}{skill_note} " + f"Checks: {best.evaluation.strict_checks}. " + f"Failures: {'; '.join(best.evaluation.failures) or 'none reported'}." + ) + exc.reasoning_steps = best.reasoning_steps + exc.conversation_id = best.conversation_id + exc.response_id = best.response_id + exc.timings = item_timings + exc.detail = detail + exc.runs_passed = runs_passed + exc.runs_effective = runs_effective + raise exc + return AgenticEvalOutcome( + runs_passed=runs_passed, + runs_effective=runs_effective, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + timings=item_timings, + ) diff --git a/packages/gooddata-eval/tests/test_agentic_report_skill.py b/packages/gooddata-eval/tests/test_agentic_report_skill.py new file mode 100644 index 000000000..4073afe74 --- /dev/null +++ b/packages/gooddata-eval/tests/test_agentic_report_skill.py @@ -0,0 +1,458 @@ +# (C) 2026 GoodData Corporation. All rights reserved. +# SPDX-License-Identifier: LicenseRef-GoodData-Enterprise +import json +from typing import Any + +import pytest +from gooddata_eval.core.agentic import report_skill +from gooddata_eval.core.agentic.report_skill import ( + ReportSkillAssertionError, + _execute_single_report_run, + _validate_expectation, + _visualizations_of, + build_simulated_reply, + evaluate_agentic_report_skill, + evaluate_report_response, + run_agentic_report_skill, +) +from gooddata_eval.core.models import ChatResult + +# Shapes follow what gen-ai writes for a drafted report (composed_report.aac.json): a cover +# page, then content pages whose layout nests `column` and `row` entries down to the slots. +_REVENUE_TREND = "revenue_trend" +_RETURNS_BY_CATEGORY = "returns_by_category" +_PERIOD = {"start": "2026-01-01", "end": "2026-06-30"} + + +def _cover_page() -> dict: + return { + "id": "page1", + "kind": "cover", + "format": "widescreen", + "layout": {"column": [{"id": "coverTitle", "weight": 2, "heading": "{reportName}", "style": "h1"}]}, + } + + +def _content_page(page_id: str, *visualizations: str) -> dict: + return { + "id": page_id, + "kind": "content", + "format": "widescreen", + "layout": { + "column": [ + {"id": "pageTitle", "weight": 2, "heading": "Revenue", "style": "h1"}, + { + "weight": 9, + "row": [ + {"weight": 2, "row": [{"id": f"widget{i}", "visualization": v, "date": "date"}]} + for i, v in enumerate(visualizations, start=1) + ], + }, + ] + }, + } + + +def _report_part( + pages: list[dict] | None = None, + *, + ref: str = "report_1", + page_count: int | None = None, + period: dict | None = None, + saved_report_id: str | None = None, + base_report_id: str | None = None, + report: dict | None | str = "default", +) -> dict: + pages = pages if pages is not None else [_cover_page(), _content_page("page2", _REVENUE_TREND)] + document = ( + { + "id": "sales_overview", + "type": "report", + "title": "Sales overview", + "period": period if period is not None else _PERIOD, + "pages": pages, + } + if report == "default" + else report + ) + return { + "type": "report", + "report_ref": ref, + "format": "aac-v1", + "report": document, + "page_count": len(pages) if page_count is None else page_count, + "base_report_id": base_report_id, + "saved_report_id": saved_report_id, + } + + +def _draft_result(*, ref: str = "report_1", page_count: int = 2, summaries_from_data: int = 1) -> dict: + return { + "status": "success", + "ref": ref, + "page_count": page_count, + "page_ids": [f"page{i}" for i in range(1, page_count + 1)], + "pages_kept": 0, + "visualization_count": 1, + "summaries_from_data": summaries_from_data, + "summaries_stale": 0, + "layouts_used": ["auto"] * page_count, + } + + +def _tool_call(name: str, result: dict | None = None) -> dict: + return {"functionName": name, "functionArguments": "{}", "result": None if result is None else json.dumps(result)} + + +def _set_skills(*skills: str) -> dict: + return _tool_call("set_skills", {"skills_to_activate": list(skills)}) + + +def _chat_result( + *, tool_calls: list[dict] | None = None, parts: list[dict] | None = None, text: str = "" +) -> ChatResult: + return ChatResult.model_validate( + { + "textResponse": text, + "toolCallEvents": tool_calls or [], + "unhandledParts": parts or [], + "streamEnded": True, + } + ) + + +def _drafting_turn(**part_kwargs: Any) -> ChatResult: + return _chat_result( + tool_calls=[_set_skills("report_builder"), _tool_call("draft_report", _draft_result())], + parts=[{"type": "text", "text": "I've put together a report with 2 slides."}, _report_part(**part_kwargs)], + text="I've put together a report with 2 slides.", + ) + + +class _ScriptedChatClient: + """Stands in for ChatClient: answers each message with the next scripted turn.""" + + def __init__(self, turns: list[ChatResult]) -> None: + self._turns = list(turns) + self.sent: list[str] = [] + self.created: list[str] = [] + self.deleted: list[str] = [] + self.closed = False + + def create_conversation(self) -> str: + conversation_id = f"conv-{len(self.created) + 1}" + self.created.append(conversation_id) + return conversation_id + + def send_message(self, conversation_id: str, question: str) -> ChatResult: + self.sent.append(question) + return self._turns.pop(0) if self._turns else _chat_result() + + def delete_conversation(self, conversation_id: str) -> None: + self.deleted.append(conversation_id) + + def close(self) -> None: + self.closed = True + + +def _install_client(monkeypatch: pytest.MonkeyPatch, client: _ScriptedChatClient) -> None: + monkeypatch.setattr(report_skill, "ChatClient", lambda **_kwargs: client) + + +# ── scoring ───────────────────────────────────────────────────────────────── + + +def _evaluate(part: dict | None, *, expected: dict | None = None, tool: dict | None = None, skill: bool = True): + return evaluate_report_response( + _draft_result() if tool is None else tool, + part, + {} if expected is None else expected, + skill_activated=skill, + ) + + +def test_a_drafted_report_returned_in_the_chat_item_passes(): + evaluation = _evaluate(_report_part()) + assert evaluation.strict_checks == { + "report_drafted": True, + "report_part_present": True, + "report_ref_matches": True, + "report_pages_consistent": True, + "report_not_saved": True, + "report_skill_activated": True, + } + assert evaluation.strict_pass + assert evaluation.failures == [] + + +def test_period_and_charts_are_scored_only_when_the_fixture_states_them(): + expected = {"period": _PERIOD, "visualizations": [{"id": _REVENUE_TREND, "title": "Revenue trend"}]} + checks = _evaluate(_report_part(), expected=expected).strict_checks + assert checks["report_period_correct"] is True + assert checks["report_charts_matched"] is True + + +def test_no_successful_draft_fails_every_check_and_says_so(): + evaluation = evaluate_report_response(None, None, {"period": _PERIOD}, skill_activated=True) + assert evaluation.strict_checks["report_drafted"] is False + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.strict_checks["report_period_correct"] is False + assert not evaluation.strict_pass + assert evaluation.failures == ["the agent never produced a successful draft_report call"] + + +def test_a_draft_without_a_report_part_fails(): + evaluation = _evaluate(None) + assert evaluation.strict_checks["report_drafted"] is True + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.failures == ["the response carries no 'report' part"] + + +def test_a_report_part_whose_document_did_not_resolve_fails(): + evaluation = _evaluate(_report_part(report=None)) + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.failures == ["the 'report' part carries no report document (report_ref 'report_1')"] + + +def test_a_document_that_is_not_a_report_fails(): + evaluation = _evaluate(_report_part(report={"id": "x", "type": "dashboard", "pages": []})) + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.failures == ["the 'report' part carries a document of type 'dashboard', expected 'report'"] + + +def test_a_part_pointing_at_another_draft_fails(): + evaluation = _evaluate(_report_part(ref="report_2")) + assert evaluation.strict_checks["report_ref_matches"] is False + assert evaluation.failures == ["the 'report' part shows 'report_2', but draft_report returned 'report_1'"] + + +def test_a_page_count_that_disagrees_with_the_pages_fails(): + evaluation = _evaluate(_report_part(page_count=3)) + assert evaluation.strict_checks["report_pages_consistent"] is False + assert evaluation.failures == ["the report has 2 page(s), but the part says 3 and draft_report said 2"] + + +def test_a_cover_alone_is_not_a_report(): + evaluation = _evaluate(_report_part([_cover_page()]), tool=_draft_result(page_count=1)) + assert evaluation.strict_checks["report_pages_consistent"] is False + assert evaluation.failures == ["the report has 1 page(s); it needs a cover and at least one content page"] + + +def test_a_new_draft_must_not_already_be_saved(): + evaluation = _evaluate(_report_part(saved_report_id="sales_overview")) + assert evaluation.strict_checks["report_not_saved"] is False + assert evaluation.failures == ["a new draft must not be saved yet, but it reports saved_report_id 'sales_overview'"] + + +def test_a_new_draft_must_not_edit_a_saved_report(): + evaluation = _evaluate(_report_part(base_report_id="q3_review")) + assert evaluation.strict_checks["report_not_saved"] is False + assert evaluation.failures == [ + "a new draft must not edit a saved report, but it reports base_report_id 'q3_review'" + ] + + +def test_a_wrong_period_fails(): + evaluation = _evaluate( + _report_part(period={"start": "2025-07-01", "end": "2025-12-31"}), expected={"period": _PERIOD} + ) + assert evaluation.strict_checks["report_period_correct"] is False + assert evaluation.failures == [ + "the report covers 2025-07-01 to 2025-12-31, expected 2026-01-01 to 2026-06-30", + ] + + +def test_a_missing_chart_fails_and_is_named(): + expected = {"visualizations": [{"id": _RETURNS_BY_CATEGORY, "title": "Returns by category"}]} + evaluation = _evaluate(_report_part(), expected=expected) + assert evaluation.strict_checks["report_charts_matched"] is False + assert evaluation.failures == ["the report does not show chart 'returns_by_category' ('Returns by category')"] + + +def test_a_routing_miss_fails_the_skill_check_alone(): + evaluation = _evaluate(_report_part(), skill=False) + assert evaluation.strict_checks["report_skill_activated"] is False + assert [name for name, ok in evaluation.strict_checks.items() if not ok] == ["report_skill_activated"] + + +def test_visualizations_are_found_however_deep_the_layout_nests_them(): + page = _content_page("page2", _REVENUE_TREND, _RETURNS_BY_CATEGORY) + assert _visualizations_of({"pages": [_cover_page(), page]}) == {_REVENUE_TREND, _RETURNS_BY_CATEGORY} + + +# ── fixture validation ────────────────────────────────────────────────────── + + +@pytest.mark.parametrize( + ("expected", "message"), + [ + ({"period": {"start": "2026-01-01"}}, "period needs both 'start' and 'end'"), + ({"period": "H1 2026"}, "period needs both 'start' and 'end'"), + ({"visualizations": []}, "visualizations is empty"), + ({"visualizations": [{"title": "Revenue"}]}, "every visualization needs an 'id'"), + ({"expects_clarification": "yes"}, "expects_clarification must be true or false"), + ], +) +def test_an_unusable_fixture_is_rejected_before_any_request(expected, message): + with pytest.raises(ValueError, match=message): + _validate_expectation(expected) + + +def test_an_empty_fixture_is_usable(): + _validate_expectation({}) + + +# ── simulated user ────────────────────────────────────────────────────────── + + +def test_the_simulated_reply_names_the_charts_and_the_period(): + expected = { + "period": _PERIOD, + "visualizations": [ + {"id": _REVENUE_TREND, "title": "Revenue trend"}, + {"id": _RETURNS_BY_CATEGORY, "title": "Returns by category"}, + ], + } + assert build_simulated_reply(expected) == ( + "Please use these charts: Revenue trend, Returns by category. " + "Period: 2026-01-01 to 2026-06-30. " + "Anything else is up to you. Please create the report now." + ) + + +def test_the_simulated_reply_leaves_out_what_the_fixture_does_not_state(): + assert build_simulated_reply({}) == "Anything else is up to you. Please create the report now." + + +# ── conversation loop ─────────────────────────────────────────────────────── + + +def test_a_report_drafted_on_the_first_turn_ends_the_run(): + client = _ScriptedChatClient([_drafting_turn()]) + run = _execute_single_report_run(client, "conv-1", "Create a sales report for H1 2026", {}, max_iterations=4) + assert client.sent == ["Create a sales report for H1 2026"] + assert run.total_turns == 1 + assert run.asked_first is False + assert run.evaluation.strict_pass + + +def test_a_clarifying_question_is_answered_and_the_report_scored(): + expected = {"period": _PERIOD, "visualizations": [{"id": _REVENUE_TREND, "title": "Revenue trend"}]} + question = _chat_result(text="Which period should the report cover?") + client = _ScriptedChatClient([question, _drafting_turn()]) + run = _execute_single_report_run(client, "conv-1", "Make me a report", expected, max_iterations=4) + assert client.sent == ["Make me a report", build_simulated_reply(expected)] + assert run.total_turns == 2 + assert run.asked_first is True + assert run.evaluation.strict_pass + + +def test_a_silent_turn_ends_the_run_without_replying(): + client = _ScriptedChatClient([_chat_result()]) + run = _execute_single_report_run(client, "conv-1", "Make me a report", {}, max_iterations=4) + assert client.sent == ["Make me a report"] + assert not run.evaluation.strict_pass + + +def test_a_copilot_that_never_drafts_stops_at_the_turn_limit(): + client = _ScriptedChatClient([_chat_result(text="Which period?")] * 5) + run = _execute_single_report_run(client, "conv-1", "Make me a report", {}, max_iterations=3) + assert len(client.sent) == 3 + assert run.evaluation.strict_checks["report_drafted"] is False + + +def test_asking_first_is_recorded_only_when_the_fixture_expects_it(): + plain = _execute_single_report_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", {}, max_iterations=4) + assert "report_asked_first" not in plain.diagnostics + + expected = {"expects_clarification": True} + run = _execute_single_report_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", expected, max_iterations=4) + assert run.diagnostics == {"report_asked_first": False} + assert run.evaluation.strict_pass, "drafting without asking is recorded, never failed" + + +# ── K runs and the gate ───────────────────────────────────────────────────── + + +def test_every_conversation_the_run_creates_is_deleted(monkeypatch): + client = _ScriptedChatClient([_drafting_turn(), _drafting_turn()]) + _install_client(monkeypatch, client) + summary = run_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, k=2) + assert summary.pass_at_k and summary.pass_power_k + assert client.deleted == client.created == ["conv-1", "conv-2"] + assert client.closed + + +def test_a_conversation_handed_in_is_not_deleted(monkeypatch): + client = _ScriptedChatClient([_drafting_turn()]) + _install_client(monkeypatch, client) + run_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, initial_conversation_id="theirs") + assert client.deleted == [] + + +def test_a_passing_item_returns_its_checks_and_draft_facts(monkeypatch): + _install_client(monkeypatch, _ScriptedChatClient([_drafting_turn()])) + outcome = evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {"period": _PERIOD}) + assert outcome.runs_passed == 1 + assert outcome.detail["report_period_correct"] is True + assert outcome.detail["summaries_from_data"] == 1 + assert outcome.detail["failures"] == [] + + +def test_a_failing_item_raises_with_the_failures(monkeypatch): + _install_client(monkeypatch, _ScriptedChatClient([_chat_result(text="I can't do that.")])) + with pytest.raises(ReportSkillAssertionError) as raised: + evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, max_iterations=1) + assert "the agent never produced a successful draft_report call" in str(raised.value) + assert raised.value.runs_passed == 0 + assert raised.value.conversation_id == "conv-1" + + +def test_a_failing_item_without_report_builder_points_at_the_flags(monkeypatch): + turn = _chat_result(tool_calls=[_set_skills("visualization")], text="Here is a chart.") + _install_client(monkeypatch, _ScriptedChatClient([turn])) + with pytest.raises(ReportSkillAssertionError, match="enableBusinessBriefingReportsApp"): + evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, max_iterations=1) + + +def test_an_unusable_fixture_fails_before_any_request(monkeypatch): + client = _ScriptedChatClient([]) + _install_client(monkeypatch, client) + with pytest.raises(ValueError): + evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {"visualizations": []}) + assert client.created == [] + + +def test_a_ref_missing_on_both_sides_does_not_match(): + evaluation = _evaluate(_report_part(ref=None), tool={**_draft_result(), "ref": None}) + assert evaluation.strict_checks["report_ref_matches"] is False + + +def test_the_last_successful_draft_of_a_turn_is_the_one_scored(): + turn = _chat_result( + tool_calls=[ + _tool_call("draft_report", _draft_result(ref="report_1")), + _tool_call("draft_report", {"status": "error", "message": "no layout fits 7 charts"}), + _tool_call("draft_report", _draft_result(ref="report_2")), + ], + parts=[_report_part(ref="report_2")], + text="I've put together a report with 2 slides.", + ) + run = _execute_single_report_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) + assert run.tool_result is not None and run.tool_result["ref"] == "report_2" + assert run.evaluation.strict_pass + + +def test_a_first_turn_that_did_not_ask_is_not_recorded_as_asking(): + refusal = _chat_result(text="I could not find any charts on that topic.") + expected = {"expects_clarification": True} + run = _execute_single_report_run(_ScriptedChatClient([refusal, _drafting_turn()]), "c", "q", expected, 4) + assert run.evaluation.strict_pass + assert run.diagnostics == {"report_asked_first": False} + + +def test_a_structured_clarifying_question_counts_as_asking(): + question = _chat_result(parts=[{"type": "clarifyingQuestions", "questions": []}]) + expected = {"expects_clarification": True} + run = _execute_single_report_run(_ScriptedChatClient([question, _drafting_turn()]), "c", "q", expected, 4) + assert run.diagnostics == {"report_asked_first": True} diff --git a/packages/gooddata-eval/tests/test_agentic_runner.py b/packages/gooddata-eval/tests/test_agentic_runner.py index 28ce3a4c1..a3e0ffc7e 100644 --- a/packages/gooddata-eval/tests/test_agentic_runner.py +++ b/packages/gooddata-eval/tests/test_agentic_runner.py @@ -85,6 +85,7 @@ def test_dispatch_agentic_omits_agent_id_by_default(): {"type": "dashboard", "visualizations": [], "date_range": None, "min_new_visualizations": 0}, "evaluate_agentic_dashboard_skill", ), + ("agentic_report_skill", {}, "evaluate_agentic_report_skill"), ("agentic_search", {"tool_call": {"function_arguments": {}}}, "evaluate_agentic_search_tool"), ("agentic_general_question", "What is X?", "evaluate_agentic_general_question"), ("agentic_guardrail", "Ignore prior instructions", "evaluate_agentic_guardrail"), diff --git a/packages/gooddata-eval/tests/test_trace_linker.py b/packages/gooddata-eval/tests/test_trace_linker.py index 2801e349f..0f15d5073 100644 --- a/packages/gooddata-eval/tests/test_trace_linker.py +++ b/packages/gooddata-eval/tests/test_trace_linker.py @@ -115,6 +115,7 @@ def test_run_trace_link_inline_runs_the_task_on_the_calling_thread(): ("metric_skill", "evaluate_agentic_metric_skill"), ("alert_skill", "evaluate_agentic_alert_skill"), ("dashboard_skill", "evaluate_agentic_dashboard_skill"), + ("report_skill", "evaluate_agentic_report_skill"), ("search_tool", "evaluate_agentic_search_tool"), ("visualization", "evaluate_agentic_visualization"), ("kda_skill", "evaluate_agentic_kda_skill"),