From 4fa39497d4934c666350d11f0ade681be06471db Mon Sep 17 00:00:00 2001 From: Roman Rakus Date: Mon, 5 Oct 2026 11:40:45 +0200 Subject: [PATCH 1/4] feat(gooddata-eval): recognize the report answer part The Report copilot answers with a `report` multipart part carrying the drafted report as code. The SSE client did not list the type, so every report turn logged an unknown-part warning. It is now a known type and, like `dashboard`, stays in `unhandled_parts` verbatim for an evaluator to read back by type. jira: LX-3174 risk: nonprod Co-Authored-By: Claude Opus 5.5 (1M context) --- .../src/gooddata_eval/core/chat/sse_client.py | 1 + .../gooddata-eval/tests/test_chat_render.py | 25 ++++++++++++++++++- 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py index bf2d602b2..a39d9fda5 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py @@ -50,6 +50,7 @@ "visualization", "dashboard", "dashboardPatch", + "report", "kda", "whatIf", "searchResults", diff --git a/packages/gooddata-eval/tests/test_chat_render.py b/packages/gooddata-eval/tests/test_chat_render.py index cebaa83d2..d8df530bf 100644 --- a/packages/gooddata-eval/tests/test_chat_render.py +++ b/packages/gooddata-eval/tests/test_chat_render.py @@ -52,6 +52,28 @@ def test_an_unmodelled_part_type_is_kept_not_dropped(): assert "kda" in render_answer_text(result) +def test_a_report_part_is_kept_verbatim_without_an_unknown_type_warning(caplog: pytest.LogCaptureFixture) -> None: + part = { + "type": "report", + "report_ref": "report_1", + "format": "aac-v1", + "report": { + "id": "sales_overview", + "type": "report", + "title": "Sales overview", + "period": {"start": "2026-01-01", "end": "2026-06-30"}, + "pages": [{"id": "page_1", "kind": "cover", "format": "16:9", "layout": {"slots": []}}], + }, + "page_count": 1, + "base_report_id": None, + "saved_report_id": None, + } + with caplog.at_level("WARNING", logger="gooddata_eval.core.chat.sse_client"): + result = parse_sse_lines(_multipart_lines({"type": "text", "text": "I've put together a report."}, part)) + assert result.unhandled_parts == [part] + assert "unknown multipart part type" not in caplog.text + + def test_an_unresolved_visualization_part_is_not_treated_as_content(): result = parse_sse_lines(_multipart_lines({"type": "visualization", "visualization": None})) assert result.unhandled_parts == [] @@ -70,6 +92,7 @@ def test_known_part_types_matches_the_documented_gen_ai_union(): "visualization", "dashboard", "dashboardPatch", + "report", "kda", "whatIf", "searchResults", @@ -112,7 +135,7 @@ def test_a_huge_unmodelled_part_is_truncated(): assert len(rendered) < 3_000 -@pytest.mark.parametrize("ptype", ["dashboard", "dashboardPatch", "kda", "whatIf", "clarifyingQuestions"]) +@pytest.mark.parametrize("ptype", ["dashboard", "dashboardPatch", "report", "kda", "whatIf", "clarifyingQuestions"]) def test_no_known_part_type_is_silently_dropped(ptype): result = parse_sse_lines(_multipart_lines({"type": ptype, "payload": {"a": 1}})) assert result.unhandled_parts, f"{ptype} was dropped" From e04c1a21b00de23900654427b83ad2936551bd73 Mon Sep 17 00:00:00 2001 From: Roman Rakus Date: Mon, 5 Oct 2026 11:50:50 +0200 Subject: [PATCH 2/4] test(gooddata-eval): use the real report document in the report-part test The fixture invented a page shape. It now follows what gen-ai writes (composed_report.aac.json): format "widescreen" and a "column" layout. jira: LX-3174 risk: nonprod Co-Authored-By: Claude Opus 5.5 (1M context) --- .../gooddata-eval/tests/test_chat_render.py | 22 +++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/packages/gooddata-eval/tests/test_chat_render.py b/packages/gooddata-eval/tests/test_chat_render.py index d8df530bf..1baad5a6b 100644 --- a/packages/gooddata-eval/tests/test_chat_render.py +++ b/packages/gooddata-eval/tests/test_chat_render.py @@ -62,9 +62,27 @@ def test_a_report_part_is_kept_verbatim_without_an_unknown_type_warning(caplog: "type": "report", "title": "Sales overview", "period": {"start": "2026-01-01", "end": "2026-06-30"}, - "pages": [{"id": "page_1", "kind": "cover", "format": "16:9", "layout": {"slots": []}}], + "pages": [ + { + "id": "page1", + "kind": "cover", + "format": "widescreen", + "layout": {"column": [{"id": "coverTitle", "weight": 2, "heading": "{reportName}", "style": "h1"}]}, + }, + { + "id": "page2", + "kind": "content", + "format": "widescreen", + "layout": { + "column": [ + {"id": "pageTitle", "weight": 2, "heading": "Revenue", "style": "h1"}, + {"id": "widget1", "weight": 9, "visualization": "revenue_trend", "date": "date"}, + ] + }, + }, + ], }, - "page_count": 1, + "page_count": 2, "base_report_id": None, "saved_report_id": None, } From aaba134906a0c6bbb2c9d84c72dc98aee337be9e Mon Sep 17 00:00:00 2001 From: Roman Rakus Date: Mon, 5 Oct 2026 13:18:19 +0200 Subject: [PATCH 3/4] feat(gooddata-eval): add the agentic report-skill evaluator Scores whether asking the chat for a report returns one. The Report copilot keeps its draft in conversation state and saves nothing, so the evaluator reads the reply only: a successful draft_report call, a `report` part carrying the report document, its ref matching the draft's, a page count that agrees with the pages (a cover plus at least one content page), and a draft that is neither saved nor editing a saved report. A fixture may also state the period and the charts the report must show; those checks run only when it does. When the copilot asks back instead of drafting, a fixed reply built from the fixture answers it, as the dashboard skill does. Whether it asked first is recorded when the fixture expects a question, never gated: how much the copilot should ask is still an open product decision. Registered as agentic_report_skill; it runs serially until the dataset has runs behind it. jira: LX-3175 risk: nonprod Co-Authored-By: Claude Opus 5.5 (1M context) --- .../src/gooddata_eval/cli/agentic_runner.py | 17 +- .../gooddata_eval/core/agentic/__init__.py | 14 + .../core/agentic/report_skill.py | 657 ++++++++++++++++++ .../tests/test_agentic_report_skill.py | 618 ++++++++++++++++ .../tests/test_agentic_runner.py | 1 + .../gooddata-eval/tests/test_trace_linker.py | 1 + 6 files changed, 1307 insertions(+), 1 deletion(-) create mode 100644 packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py create mode 100644 packages/gooddata-eval/tests/test_agentic_report_skill.py diff --git a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py index 20885a85c..87bece139 100644 --- a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py @@ -18,6 +18,7 @@ from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill +from gooddata_eval.core.agentic.report_skill import evaluate_agentic_report_skill from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization from gooddata_eval.core.agentic.what_if import evaluate_agentic_what_if @@ -43,6 +44,7 @@ class _LfKw(TypedDict, total=False): "agentic_metric_skill", "agentic_alert_skill", "agentic_dashboard_skill", + "agentic_report_skill", "agentic_search", "agentic_general_question", "agentic_guardrail", @@ -92,7 +94,8 @@ class _LfKw(TypedDict, total=False): # # agentic_dashboard_skill is absent by default rather than by evidence: gen-ai holds the draft and # any chart it authors in conversation state and writes neither until a user saves from the UI, so -# it is a candidate for the allowlist once the dataset has runs behind it. +# it is a candidate for the allowlist once the dataset has runs behind it. agentic_report_skill is +# absent for the same reason: the report draft stays in conversation state until a user saves it. WORKSPACE_MUTATING_TEST_KINDS = frozenset(AGENTIC_TEST_KINDS) - PARALLEL_SAFE_TEST_KINDS @@ -203,6 +206,18 @@ def _dispatch_agentic( agent_id=agent_id, **lf_kw, ) + elif kind == "agentic_report_skill": + return evaluate_agentic_report_skill( + host=host, + token=token, + workspace_id=workspace_id, + question=item.question, + expected_output=eo, + k=k, + gate=gate, + agent_id=agent_id, + **lf_kw, + ) elif kind == "agentic_alert_skill": return evaluate_agentic_alert_skill( host=host, diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py index 3d74a4b9f..9ddbd1b20 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py @@ -53,6 +53,14 @@ evaluate_agentic_metric_skill, run_agentic_metric_skill, ) +from gooddata_eval.core.agentic.report_skill import ( + AgenticReportSummary, + ReportEvaluation, + ReportRunResult, + ReportSkillAssertionError, + evaluate_agentic_report_skill, + run_agentic_report_skill, +) from gooddata_eval.core.agentic.search_tool import ( AgenticSearchSummary, SearchResult, @@ -75,6 +83,7 @@ "AgenticGuardrailSummary", "AgenticKdaSummary", "AgenticMetricSummary", + "AgenticReportSummary", "AgenticSearchSummary", "AgenticRunSummary", "AlertEvaluation", @@ -95,6 +104,9 @@ "KdaSkillAssertionError", "MetricRunResult", "MetricSkillAssertionError", + "ReportEvaluation", + "ReportRunResult", + "ReportSkillAssertionError", "RunResult", "SearchResult", "SearchToolAssertionError", @@ -108,6 +120,7 @@ "evaluate_agentic_guardrail", "evaluate_agentic_kda_skill", "evaluate_agentic_metric_skill", + "evaluate_agentic_report_skill", "evaluate_agentic_search_tool", "evaluate_agentic_visualization", "run_agentic_alert_skill", @@ -117,6 +130,7 @@ "run_agentic_guardrail", "run_agentic_kda_skill", "run_agentic_metric_skill", + "run_agentic_report_skill", "run_agentic_search_tool", "run_agentic_visualization", ] diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py new file mode 100644 index 000000000..e753f58e2 --- /dev/null +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py @@ -0,0 +1,657 @@ +# (C) 2026 GoodData Corporation. All rights reserved. +"""Agentic report-skill evaluation runner.""" + +from __future__ import annotations + +import time +from collections.abc import Iterator +from dataclasses import dataclass, field +from typing import Any + +from gooddata_eval.core.agentic._conversation_context import classify_reply +from gooddata_eval.core.agentic._gate import ( + DEFAULT_GATE, + EvalGate, + gate_failure_note, + gate_passed, + log_gate_scores, + stamp_gate_metadata, +) +from gooddata_eval.core.agentic._trace_linker import ( + RunIdentity, + RunTraceContext, + SubmitTraceLink, + open_trace_window, + run_trace_link_inline, + submit_trace_scoring, + utc_now, +) +from gooddata_eval.core.agentic.dashboard_skill import _extract_tool_result, _skill_activated +from gooddata_eval.core.chat.render import render_answer_text +from gooddata_eval.core.chat.sse_client import ChatClient +from gooddata_eval.core.config import ReasoningEffort +from gooddata_eval.core.models import ( + AgenticAssertionError, + AgenticEvalOutcome, + ChatResult, + ReasoningStepEvent, + ToolCallEvent, + build_latency_breakdown, + shift_and_index_events, +) +from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings + +_DEFAULT_K = 1 +# Same slack as the dashboard skill: the simulated reply is fixed, so the extra rounds only +# absorb a turn that answered without drafting. +_DEFAULT_MAX_ITERATIONS = 4 + +_DRAFT_TOOL = "draft_report" +_BUILDER_SKILL = "report_builder" +_PART_TYPE = "report" +# The copilot's own rule: a report opens with a cover and has at least one content page. +_COVER_PAGE = "cover" +_CONTENT_PAGE = "content" +_EXPECTATION_KEYS = frozenset({"period", "visualizations", "expects_clarification"}) + + +def _extract_report_part(chat_result: ChatResult) -> dict | None: + """The last ``report`` part of the turn, read back from ``unhandled_parts`` by type. + + The last one, because a turn that drafts and then refines leaves the refined version as + the one the user sees. + """ + for part in reversed(chat_result.unhandled_parts): + if isinstance(part, dict) and part.get("type") == _PART_TYPE: + return part + return None + + +def _nodes(node: Any) -> Iterator[dict]: + """Every node of a page layout in document order, following ``row`` and ``column`` as gen-ai does.""" + if not isinstance(node, dict): + return + yield node + for direction in ("row", "column"): + for child in node.get(direction) or []: + yield from _nodes(child) + + +def _page_kind(page: dict) -> str: + """A page's kind; gen-ai reads a page without one as a content page.""" + return str(page.get("kind") or _CONTENT_PAGE) + + +def _visualizations_of(report: dict) -> set[str]: + """Every visualization id placed anywhere in the report's page layouts.""" + return { + node["visualization"] + for page in report.get("pages") or [] + if isinstance(page, dict) + for node in _nodes(page.get("layout")) + if isinstance(node.get("visualization"), str) + } + + +def _has_period(expected_output: dict) -> bool: + return "period" in expected_output + + +def _has_visualizations(expected_output: dict) -> bool: + return "visualizations" in expected_output + + +def _expects_clarification(expected_output: dict) -> bool: + return "expects_clarification" in expected_output + + +def _validate_expectation(expected_output: Any) -> None: + """Reject a fixture the run could not score meaningfully, before the first API call. + + Every key is optional. A key that is present has to be usable, because a malformed one + would otherwise score vacuously or only surface on the branch where the copilot asks back. + An unknown key is rejected too: a misspelt ``visualisations`` would otherwise leave the + item scored on structure alone. + + Raises: + ValueError: the expectation is unusable. + """ + if not isinstance(expected_output, dict): + raise ValueError(f"expected_output must be an object, got {type(expected_output).__name__}") + unknown = sorted(set(expected_output) - _EXPECTATION_KEYS) + if unknown: + raise ValueError(f"unknown expected_output key(s) {unknown}; known: {sorted(_EXPECTATION_KEYS)}") + if _has_period(expected_output): + period = expected_output.get("period") + if not isinstance(period, dict) or not period.get("start") or not period.get("end"): + raise ValueError(f"period needs both 'start' and 'end', got {period!r}") + if _has_visualizations(expected_output): + visualizations = expected_output.get("visualizations") + if not isinstance(visualizations, list) or not visualizations: + raise ValueError("visualizations is empty; the chart check would pass vacuously") + for entry in visualizations: + if not isinstance(entry, dict) or not entry.get("id"): + raise ValueError(f"every visualization needs an 'id', got {entry!r}") + if _expects_clarification(expected_output) and not isinstance(expected_output["expects_clarification"], bool): + raise ValueError( + f"expects_clarification must be true or false, got {expected_output['expects_clarification']!r}" + ) + + +def build_simulated_reply(expected_output: dict) -> str: + """The reply the simulated user sends when the copilot asks back instead of drafting. + + Deterministic and LLM-free, built from what the fixture states, so a failure stays the + copilot's rather than a simulated user that phrased things differently each run. + """ + segments: list[str] = [] + visualizations = expected_output.get("visualizations") or [] + if visualizations: + titles = ", ".join(str(v.get("title") or v.get("id")) for v in visualizations) + segments.append(f"Please use these charts: {titles}.") + period = expected_output.get("period") + if isinstance(period, dict): + segments.append(f"Period: {period.get('start')} to {period.get('end')}.") + segments.append("Anything else is up to you. Please create the report now.") + return " ".join(segments) + + +@dataclass(frozen=True) +class _Applies: + """Which conditional checks the case applies; published only when they do. + + A check that could not fail is not evidence, and publishing it as passed would lift + ``quality_score`` above what the run earned. + """ + + period: bool + charts: bool + + +@dataclass +class ReportEvaluation: + """Per-run outcome of the report-skill checks; ``strict_checks`` is what the run is scored on.""" + + drafted: bool + part_present: bool + ref_matches: bool + pages_consistent: bool + not_saved: bool + skill_activated: bool + applies: _Applies + period_correct: bool = False + charts_matched: bool = False + failures: list[str] = field(default_factory=list) + + @property + def strict_pass(self) -> bool: + return all(self.strict_checks.values()) + + @property + def strict_checks(self) -> dict[str, bool]: + # Every name carries the `report_` prefix so a trace can be told apart from other skills' + # by its score names alone; a name shared with another skill would make that ambiguous. + checks = { + "report_drafted": self.drafted, + "report_part_present": self.part_present, + "report_ref_matches": self.ref_matches, + "report_pages_consistent": self.pages_consistent, + "report_not_saved": self.not_saved, + "report_skill_activated": self.skill_activated, + } + if self.applies.period: + checks["report_period_correct"] = self.period_correct + if self.applies.charts: + checks["report_charts_matched"] = self.charts_matched + return checks + + +def _read_report(report_part: dict | None) -> tuple[dict | None, str | None]: + """The part's report document, or why it carries no usable one.""" + if report_part is None: + return None, f"the response carries no {_PART_TYPE!r} part" + report = report_part.get("report") + if not isinstance(report, dict): + return ( + None, + f"the {_PART_TYPE!r} part carries no report document (report_ref {report_part.get('report_ref')!r})", + ) + if report.get("type") != _PART_TYPE: + return None, ( + f"the {_PART_TYPE!r} part carries a document of type {report.get('type')!r}, expected {_PART_TYPE!r}" + ) + return report, None + + +def evaluate_report_response( + tool_result: dict | None, + report_part: dict | None, + expected_output: dict, + *, + skill_activated: bool, +) -> ReportEvaluation: + """Score one report response against its expectation. + + Pure: no network and no conversation state, so the whole assertion surface is unit-testable + without an agent. + """ + applies = _Applies(period=_has_period(expected_output), charts=_has_visualizations(expected_output)) + + if tool_result is None: + return ReportEvaluation( + drafted=False, + part_present=False, + ref_matches=False, + pages_consistent=False, + not_saved=False, + skill_activated=skill_activated, + applies=applies, + failures=[f"the agent never produced a successful {_DRAFT_TOOL} call"], + ) + + report, part_failure = _read_report(report_part) + if report is None or report_part is None: + return ReportEvaluation( + drafted=True, + part_present=False, + ref_matches=False, + pages_consistent=False, + not_saved=False, + skill_activated=skill_activated, + applies=applies, + failures=[part_failure] if part_failure else [], + ) + + failures: list[str] = [] + + part_ref, tool_ref = report_part.get("report_ref"), tool_result.get("ref") + ref_matches = isinstance(part_ref, str) and bool(part_ref) and part_ref == tool_ref + if not ref_matches: + failures.append(f"the {_PART_TYPE!r} part shows {part_ref!r}, but {_DRAFT_TOOL} returned {tool_ref!r}") + + pages = report.get("pages") or [] + part_count, tool_count = report_part.get("page_count"), tool_result.get("page_count") + if part_count != len(pages) or tool_count != len(pages): + failures.append( + f"the report has {len(pages)} page(s), but the part says {part_count} and {_DRAFT_TOOL} said {tool_count}" + ) + pages_consistent = False + else: + kinds = [_page_kind(page) for page in pages if isinstance(page, dict)] + if not kinds or kinds[0] != _COVER_PAGE: + failures.append(f"the report opens with a {kinds[0] if kinds else None!r} page, not a cover") + if _CONTENT_PAGE not in kinds: + failures.append("the report has no content page") + pages_consistent = bool(kinds) and kinds[0] == _COVER_PAGE and _CONTENT_PAGE in kinds + + not_saved = True + for key, wording in (("saved_report_id", "be saved yet"), ("base_report_id", "edit a saved report")): + value = report_part.get(key) + if value is not None: + failures.append(f"a new draft must not {wording}, but it reports {key} {value!r}") + not_saved = False + + period_correct = False + if applies.period: + expected_period = expected_output["period"] + actual_period = report.get("period") or {} + period_correct = all(actual_period.get(key) == expected_period.get(key) for key in ("start", "end")) + if not period_correct: + failures.append( + f"the report covers {actual_period.get('start')} to {actual_period.get('end')}, " + f"expected {expected_period.get('start')} to {expected_period.get('end')}" + ) + + charts_matched = False + if applies.charts: + placed = _visualizations_of(report) + missing = [v for v in expected_output["visualizations"] if v.get("id") not in placed] + failures.extend(f"the report does not show chart {v.get('id')!r} ({v.get('title')!r})" for v in missing) + charts_matched = not missing + + return ReportEvaluation( + drafted=True, + part_present=True, + ref_matches=ref_matches, + pages_consistent=pages_consistent, + not_saved=not_saved, + skill_activated=skill_activated, + applies=applies, + period_correct=period_correct, + charts_matched=charts_matched, + failures=failures, + ) + + +@dataclass +class ReportRunResult: + """Outcome of one conversation.""" + + conversation_id: str + evaluation: ReportEvaluation + expects_clarification: bool = False + asked_first: bool = False + tool_result: dict | None = None + report_part: dict | None = None + total_turns: int = 0 + total_steps: int = 0 + reasoning_steps: list[str] = field(default_factory=list) + response_id: str | None = None + tool_call_events: list[ToolCallEvent] = field(default_factory=list) + reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + timings: PhaseTimings = field(default_factory=PhaseTimings) + + @property + def diagnostics(self) -> dict[str, bool]: + """Observed but not scored. + + Whether the copilot asked before drafting is recorded only when the fixture says it + expects a question, and never gates: how much the copilot should ask is a product + decision, not something a run can get wrong. + """ + return {"report_asked_first": self.asked_first} if self.expects_clarification else {} + + @property + def summaries_from_data(self) -> int | None: + value = (self.tool_result or {}).get("summaries_from_data") + return value if isinstance(value, int) else None + + +@dataclass +class AgenticReportSummary: + """Aggregated outcome of K runs.""" + + run_results: list[ReportRunResult] + pass_at_k: bool + pass_power_k: bool + best: ReportRunResult + + +def _execute_single_report_run( + client: ChatClient, + conversation_id: str, + question: str, + expected_output: dict, + max_iterations: int, +) -> ReportRunResult: + """Drive one conversation until the copilot drafts a report, then evaluate it. + + Nothing is cleaned up on the way out by design: gen-ai keeps the draft in conversation + state and persists nothing until a user saves it. + """ + tool_result: dict | None = None + report_part: dict | None = None + turns = 0 + steps = 0 + current_question = question + reasoning_steps: list[str] = [] + response_id: str | None = None + all_tool_call_events: list[ToolCallEvent] = [] + all_reasoning_step_events: list[ReasoningStepEvent] = [] + timings = PhaseTimings() + turn_offset = 0.0 + tool_index_offset = 0 + reasoning_index_offset = 0 + first_turn_asked = False + + for iteration in range(max_iterations): + turns += 1 + agent_started = time.monotonic() + chat_result = client.send_message(conversation_id, current_question) + agent_elapsed = time.monotonic() - agent_started + timings.agent_s += agent_elapsed + reasoning_steps.extend(chat_result.reasoning_steps or []) + response_id = chat_result.response_id or response_id + turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events( + chat_result, + turn_offset=turn_offset, + tool_index_offset=tool_index_offset, + reasoning_index_offset=reasoning_index_offset, + ) + all_tool_call_events.extend(chat_result.tool_call_events or []) + all_reasoning_step_events.extend(chat_result.reasoning_step_events or []) + steps += chat_result.reasoning_step_count + + candidate = _extract_tool_result(chat_result.tool_call_events or [], _DRAFT_TOOL) + if candidate is not None: + log_timer( + f"[timer] report_skill {conversation_id} GoodData turn {turns} complete after " + f"{agent_elapsed:.2f}s; {_DRAFT_TOOL} result received" + ) + tool_result = candidate + # Read from the same turn as the draft: the part shows the version that call stored. + report_part = _extract_report_part(chat_result) + break + + response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) + if iteration == 0: + first_turn_asked = classify_reply(chat_result, response_text) == "question" + if not response_text and not chat_result.tool_call_events: + break + if iteration >= max_iterations - 1: + break + log_timer( + f"[timer] report_skill {conversation_id} GoodData turn {turns} complete after " + f"{agent_elapsed:.2f}s; answering with the expected charts and period" + ) + current_question = build_simulated_reply(expected_output) + + return ReportRunResult( + conversation_id=conversation_id, + evaluation=evaluate_report_response( + tool_result, + report_part, + expected_output, + skill_activated=_skill_activated(all_tool_call_events, _BUILDER_SKILL), + ), + expects_clarification=_expects_clarification(expected_output), + asked_first=first_turn_asked, + tool_result=tool_result, + report_part=report_part, + total_turns=turns, + total_steps=steps, + reasoning_steps=reasoning_steps, + response_id=response_id, + tool_call_events=all_tool_call_events, + reasoning_step_events=all_reasoning_step_events, + timings=timings, + ) + + +def run_agentic_report_skill( + host: str, + token: str, + workspace_id: str, + question: str, + expected_output: dict, + k: int = _DEFAULT_K, + max_iterations: int = _DEFAULT_MAX_ITERATIONS, + initial_conversation_id: str | None = None, + reasoning_effort: ReasoningEffort | None = None, + agent_id: str | None = None, +) -> AgenticReportSummary: + """Run the report-skill agentic evaluation K times and return a summary. + + Raises: + ValueError: the fixture is unusable — see ``_validate_expectation``. + """ + _validate_expectation(expected_output) + run_results: list[ReportRunResult] = [] + client = ChatClient( + host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + ) + + try: + conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() + try: + run_results.append(_execute_single_report_run(client, conv_id_0, question, expected_output, max_iterations)) + finally: + if initial_conversation_id is None: # only delete conversations we created + client.delete_conversation(conv_id_0) + + for _ in range(1, k): + conv_id = client.create_conversation() + try: + run_results.append( + _execute_single_report_run(client, conv_id, question, expected_output, max_iterations) + ) + finally: + client.delete_conversation(conv_id) + finally: + client.close() + + return AgenticReportSummary( + run_results=run_results, + pass_at_k=any(r.evaluation.strict_pass for r in run_results), + pass_power_k=all(r.evaluation.strict_pass for r in run_results), + best=max(run_results, key=lambda r: sum(r.evaluation.strict_checks.values())), + ) + + +class ReportSkillAssertionError(AgenticAssertionError): + """Raised when a report-skill evaluation fails.""" + + +def evaluate_agentic_report_skill( + host: str, + token: str, + workspace_id: str, + question: str, + expected_output: dict, + k: int = _DEFAULT_K, + max_iterations: int = _DEFAULT_MAX_ITERATIONS, + initial_conversation_id: str | None = None, + agent_id: str | None = None, + langfuse: object | None = None, + dataset_item_id: str = "", + dataset_name: str = "report_skill", + run_timestamp: str | None = None, + model_version_override: str | None = None, + run_metadata_extra: dict | None = None, + reasoning_effort: ReasoningEffort | None = None, + submit_trace_link: SubmitTraceLink = run_trace_link_inline, + gate: EvalGate = DEFAULT_GATE, +) -> AgenticEvalOutcome: + """Run report-skill evaluation, log to Langfuse, and raise on failure. + + Returns the best run's outcome on success; on failure the same values are attached to the + raised ``ReportSkillAssertionError`` so callers can retrieve them either way. + + Raises: + ReportSkillAssertionError: the gate did not pass. + ValueError: the fixture is unusable — see ``_validate_expectation``. Raised before any + request, so it means a fixture to fix rather than a result to read. + """ + langfuse, window_start = open_trace_window(langfuse) + summary = run_agentic_report_skill( + host=host, + token=token, + workspace_id=workspace_id, + question=question, + expected_output=expected_output, + k=k, + max_iterations=max_iterations, + initial_conversation_id=initial_conversation_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + ) + + if langfuse is not None and dataset_item_id: + # Pinned on the calling thread: a deferred poll must not widen its query window. + window_end = utc_now() + + def _write_scores(ctx: RunTraceContext) -> None: + stamp_gate_metadata(ctx.run_metadata, k=len(summary.run_results), gate=gate) + + for run_idx, run in enumerate(summary.run_results): + pt = ctx.trace(run.conversation_id) + strict_checks = run.evaluation.strict_checks + with ctx.observe(pt, run_idx, conversation_id=run.conversation_id, output=strict_checks) as tid: + for score_name, value in strict_checks.items(): + ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN") + for name, value in run.diagnostics.items(): + ctx.score(tid, name=name, value=float(value), data_type="BOOLEAN") + if run.summaries_from_data is not None: + ctx.score( + tid, name="report_summaries_from_data", value=run.summaries_from_data, data_type="NUMERIC" + ) + ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC") + ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC") + log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k) + ctx.quality( + tid, + strict_checks=strict_checks, + latency_sec=pt.latency if pt else None, + cost_usd=pt.total_cost if pt else None, + ) + + # Before the pass@K raise: a failing item's scores are the ones worth having. + submit_trace_scoring( + submit_trace_link, + RunIdentity( + host, + token, + workspace_id, + dataset_name, + run_timestamp, + model_version_override, + run_metadata_extra, + reasoning_effort, + ), + langfuse=langfuse, + dataset_item_id=dataset_item_id, + conversation_ids=[r.conversation_id for r in summary.run_results], + window_start=window_start, + window_end=window_end, + suffix_runs=len(summary.run_results) > 1, + write_scores=_write_scores, + item_input=question, + ) + + item_timings = sum_timings([r.timings for r in summary.run_results]) + runs_passed = sum(1 for r in summary.run_results if r.evaluation.strict_pass) + runs_effective = len(summary.run_results) + + best = summary.best + detail: dict[str, Any] = { + **best.evaluation.strict_checks, + **best.diagnostics, + "summaries_from_data": best.summaries_from_data, + "turns": best.total_turns, + "failures": best.evaluation.failures, + "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), + } + + if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): + gate_note = gate_failure_note(gate, runs_passed, runs_effective) + skill_note = ( + "" + if best.evaluation.skill_activated + else ( + f" set_skills never activated {_BUILDER_SKILL}: either the copilot routed elsewhere, or the skill" + " is not registered. It registers only with enableGenAiReportBuilderSkill and the org's" + " enableBusinessBriefingReportsApp both on." + ) + ) + exc = ReportSkillAssertionError( + f"Report skill assertion failed. {gate_note}{skill_note} " + f"Checks: {best.evaluation.strict_checks}. " + f"Failures: {'; '.join(best.evaluation.failures) or 'none reported'}." + ) + exc.reasoning_steps = best.reasoning_steps + exc.conversation_id = best.conversation_id + exc.response_id = best.response_id + exc.timings = item_timings + exc.detail = detail + exc.runs_passed = runs_passed + exc.runs_effective = runs_effective + raise exc + return AgenticEvalOutcome( + runs_passed=runs_passed, + runs_effective=runs_effective, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + timings=item_timings, + ) diff --git a/packages/gooddata-eval/tests/test_agentic_report_skill.py b/packages/gooddata-eval/tests/test_agentic_report_skill.py new file mode 100644 index 000000000..012b68473 --- /dev/null +++ b/packages/gooddata-eval/tests/test_agentic_report_skill.py @@ -0,0 +1,618 @@ +# (C) 2026 GoodData Corporation. All rights reserved. +# SPDX-License-Identifier: LicenseRef-GoodData-Enterprise +import json +from collections.abc import Iterator +from contextlib import contextmanager +from typing import Any + +import pytest +from gooddata_eval.cli.agentic_runner import _dispatch_agentic +from gooddata_eval.core.agentic import report_skill +from gooddata_eval.core.agentic.report_skill import ( + ReportEvaluation, + ReportSkillAssertionError, + _execute_single_report_run, + _validate_expectation, + _visualizations_of, + build_simulated_reply, + evaluate_agentic_report_skill, + evaluate_report_response, + run_agentic_report_skill, +) +from gooddata_eval.core.models import ChatResult, DatasetItem + +# Shapes follow what gen-ai writes for a drafted report (composed_report.aac.json): a cover +# page, then content pages whose layout nests `column` and `row` entries down to the slots. +_REVENUE_TREND = "revenue_trend" +_RETURNS_BY_CATEGORY = "returns_by_category" +_PERIOD = {"start": "2026-01-01", "end": "2026-06-30"} + + +def _cover_page() -> dict: + return { + "id": "page1", + "kind": "cover", + "format": "widescreen", + "layout": {"column": [{"id": "coverTitle", "weight": 2, "heading": "{reportName}", "style": "h1"}]}, + } + + +def _content_page(page_id: str, *visualizations: str) -> dict: + return { + "id": page_id, + "kind": "content", + "format": "widescreen", + "layout": { + "column": [ + {"id": "pageTitle", "weight": 2, "heading": "Revenue", "style": "h1"}, + { + "weight": 9, + "row": [ + {"weight": 2, "row": [{"id": f"widget{i}", "visualization": v, "date": "date"}]} + for i, v in enumerate(visualizations, start=1) + ], + }, + ] + }, + } + + +def _report_part( + pages: list[dict] | None = None, + *, + ref: str = "report_1", + page_count: int | None = None, + period: dict | None = None, + saved_report_id: str | None = None, + base_report_id: str | None = None, + report: dict | None | str = "default", +) -> dict: + pages = pages if pages is not None else [_cover_page(), _content_page("page2", _REVENUE_TREND)] + document = ( + { + "id": "sales_overview", + "type": "report", + "title": "Sales overview", + "period": period if period is not None else _PERIOD, + "pages": pages, + } + if report == "default" + else report + ) + return { + "type": "report", + "report_ref": ref, + "format": "aac-v1", + "report": document, + "page_count": len(pages) if page_count is None else page_count, + "base_report_id": base_report_id, + "saved_report_id": saved_report_id, + } + + +def _draft_result(*, ref: str = "report_1", page_count: int = 2, summaries_from_data: int = 1) -> dict: + return { + "status": "success", + "ref": ref, + "page_count": page_count, + "page_ids": [f"page{i}" for i in range(1, page_count + 1)], + "pages_kept": 0, + "visualization_count": 1, + "summaries_from_data": summaries_from_data, + "summaries_stale": 0, + "layouts_used": ["auto"] * page_count, + } + + +def _tool_call(name: str, result: dict | None = None) -> dict: + return {"functionName": name, "functionArguments": "{}", "result": None if result is None else json.dumps(result)} + + +def _set_skills(*skills: str) -> dict: + return _tool_call("set_skills", {"skills_to_activate": list(skills)}) + + +def _chat_result( + *, tool_calls: list[dict] | None = None, parts: list[dict] | None = None, text: str = "" +) -> ChatResult: + return ChatResult.model_validate( + { + "textResponse": text, + "toolCallEvents": tool_calls or [], + "unhandledParts": parts or [], + "streamEnded": True, + } + ) + + +def _drafting_turn(**part_kwargs: Any) -> ChatResult: + return _chat_result( + tool_calls=[_set_skills("report_builder"), _tool_call("draft_report", _draft_result())], + parts=[{"type": "text", "text": "I've put together a report with 2 slides."}, _report_part(**part_kwargs)], + text="I've put together a report with 2 slides.", + ) + + +class _ScriptedChatClient: + """Stands in for ChatClient: answers each message with the next scripted turn.""" + + def __init__(self, turns: list[ChatResult]) -> None: + self._turns = list(turns) + self.sent: list[str] = [] + self.created: list[str] = [] + self.deleted: list[str] = [] + self.closed = False + + def create_conversation(self) -> str: + conversation_id = f"conv-{len(self.created) + 1}" + self.created.append(conversation_id) + return conversation_id + + def send_message(self, conversation_id: str, question: str) -> ChatResult: + self.sent.append(question) + return self._turns.pop(0) if self._turns else _chat_result() + + def delete_conversation(self, conversation_id: str) -> None: + self.deleted.append(conversation_id) + + def close(self) -> None: + self.closed = True + + +def _install_client(monkeypatch: pytest.MonkeyPatch, client: _ScriptedChatClient) -> None: + monkeypatch.setattr(report_skill, "ChatClient", lambda **_kwargs: client) + + +# ── scoring ───────────────────────────────────────────────────────────────── + + +def _evaluate( + part: dict | None, *, expected: dict | None = None, tool: dict | None = None, skill: bool = True +) -> ReportEvaluation: + return evaluate_report_response( + _draft_result() if tool is None else tool, + part, + {} if expected is None else expected, + skill_activated=skill, + ) + + +def test_a_drafted_report_returned_in_the_chat_item_passes() -> None: + evaluation = _evaluate(_report_part()) + assert evaluation.strict_checks == { + "report_drafted": True, + "report_part_present": True, + "report_ref_matches": True, + "report_pages_consistent": True, + "report_not_saved": True, + "report_skill_activated": True, + } + assert evaluation.strict_pass + assert evaluation.failures == [] + + +def test_period_and_charts_are_scored_only_when_the_fixture_states_them() -> None: + expected = {"period": _PERIOD, "visualizations": [{"id": _REVENUE_TREND, "title": "Revenue trend"}]} + checks = _evaluate(_report_part(), expected=expected).strict_checks + assert checks["report_period_correct"] is True + assert checks["report_charts_matched"] is True + + +def test_no_successful_draft_fails_every_check_and_says_so() -> None: + evaluation = evaluate_report_response(None, None, {"period": _PERIOD}, skill_activated=True) + assert evaluation.strict_checks["report_drafted"] is False + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.strict_checks["report_period_correct"] is False + assert not evaluation.strict_pass + assert evaluation.failures == ["the agent never produced a successful draft_report call"] + + +def test_a_draft_without_a_report_part_fails() -> None: + evaluation = _evaluate(None) + assert evaluation.strict_checks["report_drafted"] is True + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.failures == ["the response carries no 'report' part"] + + +def test_a_report_part_whose_document_did_not_resolve_fails() -> None: + evaluation = _evaluate(_report_part(report=None)) + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.failures == ["the 'report' part carries no report document (report_ref 'report_1')"] + + +def test_a_document_that_is_not_a_report_fails() -> None: + evaluation = _evaluate(_report_part(report={"id": "x", "type": "dashboard", "pages": []})) + assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.failures == ["the 'report' part carries a document of type 'dashboard', expected 'report'"] + + +def test_a_part_pointing_at_another_draft_fails() -> None: + evaluation = _evaluate(_report_part(ref="report_2")) + assert evaluation.strict_checks["report_ref_matches"] is False + assert evaluation.failures == ["the 'report' part shows 'report_2', but draft_report returned 'report_1'"] + + +def test_a_page_count_that_disagrees_with_the_pages_fails() -> None: + evaluation = _evaluate(_report_part(page_count=3)) + assert evaluation.strict_checks["report_pages_consistent"] is False + assert evaluation.failures == ["the report has 2 page(s), but the part says 3 and draft_report said 2"] + + +def test_a_cover_alone_is_not_a_report() -> None: + evaluation = _evaluate(_report_part([_cover_page()]), tool=_draft_result(page_count=1)) + assert evaluation.strict_checks["report_pages_consistent"] is False + assert evaluation.failures == ["the report has no content page"] + + +def test_a_new_draft_must_not_already_be_saved() -> None: + evaluation = _evaluate(_report_part(saved_report_id="sales_overview")) + assert evaluation.strict_checks["report_not_saved"] is False + assert evaluation.failures == ["a new draft must not be saved yet, but it reports saved_report_id 'sales_overview'"] + + +def test_a_new_draft_must_not_edit_a_saved_report() -> None: + evaluation = _evaluate(_report_part(base_report_id="q3_review")) + assert evaluation.strict_checks["report_not_saved"] is False + assert evaluation.failures == [ + "a new draft must not edit a saved report, but it reports base_report_id 'q3_review'" + ] + + +def test_a_wrong_period_fails() -> None: + evaluation = _evaluate( + _report_part(period={"start": "2025-07-01", "end": "2025-12-31"}), expected={"period": _PERIOD} + ) + assert evaluation.strict_checks["report_period_correct"] is False + assert evaluation.failures == [ + "the report covers 2025-07-01 to 2025-12-31, expected 2026-01-01 to 2026-06-30", + ] + + +def test_a_missing_chart_fails_and_is_named() -> None: + expected = {"visualizations": [{"id": _RETURNS_BY_CATEGORY, "title": "Returns by category"}]} + evaluation = _evaluate(_report_part(), expected=expected) + assert evaluation.strict_checks["report_charts_matched"] is False + assert evaluation.failures == ["the report does not show chart 'returns_by_category' ('Returns by category')"] + + +def test_a_routing_miss_fails_the_skill_check_alone() -> None: + evaluation = _evaluate(_report_part(), skill=False) + assert evaluation.strict_checks["report_skill_activated"] is False + assert [name for name, ok in evaluation.strict_checks.items() if not ok] == ["report_skill_activated"] + + +def test_visualizations_are_found_however_deep_the_layout_nests_them() -> None: + page = _content_page("page2", _REVENUE_TREND, _RETURNS_BY_CATEGORY) + assert _visualizations_of({"pages": [_cover_page(), page]}) == {_REVENUE_TREND, _RETURNS_BY_CATEGORY} + + +# ── fixture validation ────────────────────────────────────────────────────── + + +@pytest.mark.parametrize( + ("expected", "message"), + [ + ({"period": {"start": "2026-01-01"}}, "period needs both 'start' and 'end'"), + ({"period": "H1 2026"}, "period needs both 'start' and 'end'"), + ({"visualizations": []}, "visualizations is empty"), + ({"visualizations": [{"title": "Revenue"}]}, "every visualization needs an 'id'"), + ({"expects_clarification": "yes"}, "expects_clarification must be true or false"), + ], +) +def test_an_unusable_fixture_is_rejected_before_any_request(expected: dict, message: str) -> None: + with pytest.raises(ValueError, match=message): + _validate_expectation(expected) + + +def test_an_empty_fixture_is_usable() -> None: + _validate_expectation({}) + + +# ── simulated user ────────────────────────────────────────────────────────── + + +def test_the_simulated_reply_names_the_charts_and_the_period() -> None: + expected = { + "period": _PERIOD, + "visualizations": [ + {"id": _REVENUE_TREND, "title": "Revenue trend"}, + {"id": _RETURNS_BY_CATEGORY, "title": "Returns by category"}, + ], + } + assert build_simulated_reply(expected) == ( + "Please use these charts: Revenue trend, Returns by category. " + "Period: 2026-01-01 to 2026-06-30. " + "Anything else is up to you. Please create the report now." + ) + + +def test_the_simulated_reply_leaves_out_what_the_fixture_does_not_state() -> None: + assert build_simulated_reply({}) == "Anything else is up to you. Please create the report now." + + +# ── conversation loop ─────────────────────────────────────────────────────── + + +def test_a_report_drafted_on_the_first_turn_ends_the_run() -> None: + client = _ScriptedChatClient([_drafting_turn()]) + run = _execute_single_report_run(client, "conv-1", "Create a sales report for H1 2026", {}, max_iterations=4) + assert client.sent == ["Create a sales report for H1 2026"] + assert run.total_turns == 1 + assert run.asked_first is False + assert run.evaluation.strict_pass + + +def test_a_clarifying_question_is_answered_and_the_report_scored() -> None: + expected = {"period": _PERIOD, "visualizations": [{"id": _REVENUE_TREND, "title": "Revenue trend"}]} + question = _chat_result(text="Which period should the report cover?") + client = _ScriptedChatClient([question, _drafting_turn()]) + run = _execute_single_report_run(client, "conv-1", "Make me a report", expected, max_iterations=4) + assert client.sent == ["Make me a report", build_simulated_reply(expected)] + assert run.total_turns == 2 + assert run.asked_first is True + assert run.evaluation.strict_pass + + +def test_a_silent_turn_ends_the_run_without_replying() -> None: + client = _ScriptedChatClient([_chat_result()]) + run = _execute_single_report_run(client, "conv-1", "Make me a report", {}, max_iterations=4) + assert client.sent == ["Make me a report"] + assert not run.evaluation.strict_pass + + +def test_a_copilot_that_never_drafts_stops_at_the_turn_limit() -> None: + client = _ScriptedChatClient([_chat_result(text="Which period?")] * 5) + run = _execute_single_report_run(client, "conv-1", "Make me a report", {}, max_iterations=3) + assert len(client.sent) == 3 + assert run.evaluation.strict_checks["report_drafted"] is False + + +def test_asking_first_is_recorded_only_when_the_fixture_expects_it() -> None: + plain = _execute_single_report_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", {}, max_iterations=4) + assert "report_asked_first" not in plain.diagnostics + + expected = {"expects_clarification": True} + run = _execute_single_report_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", expected, max_iterations=4) + assert run.diagnostics == {"report_asked_first": False} + assert run.evaluation.strict_pass, "drafting without asking is recorded, never failed" + + +# ── K runs and the gate ───────────────────────────────────────────────────── + + +def test_every_conversation_the_run_creates_is_deleted(monkeypatch: pytest.MonkeyPatch) -> None: + client = _ScriptedChatClient([_drafting_turn(), _drafting_turn()]) + _install_client(monkeypatch, client) + summary = run_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, k=2) + assert summary.pass_at_k and summary.pass_power_k + assert client.deleted == client.created == ["conv-1", "conv-2"] + assert client.closed + + +def test_a_conversation_handed_in_is_not_deleted(monkeypatch: pytest.MonkeyPatch) -> None: + client = _ScriptedChatClient([_drafting_turn()]) + _install_client(monkeypatch, client) + run_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, initial_conversation_id="theirs") + assert client.deleted == [] + + +def test_a_passing_item_returns_its_checks_and_draft_facts(monkeypatch: pytest.MonkeyPatch) -> None: + _install_client(monkeypatch, _ScriptedChatClient([_drafting_turn()])) + outcome = evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {"period": _PERIOD}) + assert outcome.runs_passed == 1 + assert outcome.detail["report_period_correct"] is True + assert outcome.detail["summaries_from_data"] == 1 + assert outcome.detail["failures"] == [] + + +def test_a_failing_item_raises_with_the_failures(monkeypatch: pytest.MonkeyPatch) -> None: + _install_client(monkeypatch, _ScriptedChatClient([_chat_result(text="I can't do that.")])) + with pytest.raises(ReportSkillAssertionError) as raised: + evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, max_iterations=1) + assert "the agent never produced a successful draft_report call" in str(raised.value) + assert raised.value.runs_passed == 0 + assert raised.value.conversation_id == "conv-1" + + +def test_a_failing_item_without_report_builder_points_at_the_flags(monkeypatch: pytest.MonkeyPatch) -> None: + turn = _chat_result(tool_calls=[_set_skills("visualization")], text="Here is a chart.") + _install_client(monkeypatch, _ScriptedChatClient([turn])) + with pytest.raises( + ReportSkillAssertionError, match="enableGenAiReportBuilderSkill and the org.s enableBusinessBriefingReportsApp" + ): + evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, max_iterations=1) + + +def test_an_unusable_fixture_fails_before_any_request(monkeypatch: pytest.MonkeyPatch) -> None: + client = _ScriptedChatClient([]) + _install_client(monkeypatch, client) + with pytest.raises(ValueError): + evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {"visualizations": []}) + assert client.created == [] + + +def test_a_ref_missing_on_both_sides_does_not_match() -> None: + evaluation = _evaluate(_report_part(ref=None), tool={**_draft_result(), "ref": None}) + assert evaluation.strict_checks["report_ref_matches"] is False + + +def test_the_last_successful_draft_of_a_turn_is_the_one_scored() -> None: + turn = _chat_result( + tool_calls=[ + _tool_call("draft_report", _draft_result(ref="report_1")), + _tool_call("draft_report", {"status": "error", "message": "no layout fits 7 charts"}), + _tool_call("draft_report", _draft_result(ref="report_2")), + ], + parts=[_report_part(ref="report_2")], + text="I've put together a report with 2 slides.", + ) + run = _execute_single_report_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) + assert run.tool_result is not None and run.tool_result["ref"] == "report_2" + assert run.evaluation.strict_pass + + +def test_a_first_turn_that_did_not_ask_is_not_recorded_as_asking() -> None: + refusal = _chat_result(text="I could not find any charts on that topic.") + expected = {"expects_clarification": True} + run = _execute_single_report_run(_ScriptedChatClient([refusal, _drafting_turn()]), "c", "q", expected, 4) + assert run.evaluation.strict_pass + assert run.diagnostics == {"report_asked_first": False} + + +def test_a_structured_clarifying_question_counts_as_asking() -> None: + question = _chat_result(parts=[{"type": "clarifyingQuestions", "questions": []}]) + expected = {"expects_clarification": True} + run = _execute_single_report_run(_ScriptedChatClient([question, _drafting_turn()]), "c", "q", expected, 4) + assert run.diagnostics == {"report_asked_first": True} + + +# ── page kinds, fixture keys, asking, two drafts in one turn ─────────────── + + +def _page(page_id: str, kind: str | None) -> dict: + page = _content_page(page_id, _REVENUE_TREND) + if kind is None: + del page["kind"] + else: + page["kind"] = kind + return page + + +def test_a_report_that_does_not_open_with_a_cover_fails() -> None: + evaluation = _evaluate(_report_part([_page("page1", "content"), _page("page2", "content")])) + assert evaluation.strict_checks["report_pages_consistent"] is False + assert evaluation.failures == ["the report opens with a 'content' page, not a cover"] + + +def test_a_report_without_a_content_page_fails() -> None: + evaluation = _evaluate(_report_part([_cover_page(), _page("page2", "section")])) + assert evaluation.strict_checks["report_pages_consistent"] is False + assert evaluation.failures == ["the report has no content page"] + + +def test_a_page_without_a_kind_is_a_content_page() -> None: + assert _evaluate(_report_part([_cover_page(), _page("page2", None)])).strict_pass + + +@pytest.mark.parametrize("key", ["visualisations", "date_range", "narative"]) +def test_an_unknown_fixture_key_is_rejected(key: str) -> None: + with pytest.raises(ValueError, match=f"unknown expected_output key.*{key}"): + _validate_expectation({key: "x"}) + + +def test_an_expected_output_that_is_not_an_object_is_rejected(monkeypatch: pytest.MonkeyPatch) -> None: + client = _ScriptedChatClient([]) + _install_client(monkeypatch, client) + item = DatasetItem( + id="i", dataset_name="ds", test_kind="agentic_report_skill", question="q", expected_output="a report" + ) + with pytest.raises(ValueError, match="expected_output must be an object"): + _dispatch_agentic( + item, + host="https://h", + token="t", + workspace_id="ws", + k=1, + langfuse=None, + run_ts="r", + model_version_override=None, + ) + assert client.created == [] + + +def test_a_copilot_that_only_ever_asks_is_recorded_as_asking() -> None: + client = _ScriptedChatClient([_chat_result(text="Which period?")] * 3) + run = _execute_single_report_run(client, "c", "q", {"expects_clarification": True}, max_iterations=2) + assert run.evaluation.strict_checks["report_drafted"] is False + assert run.diagnostics == {"report_asked_first": True} + + +def test_asking_first_is_recorded_whenever_the_fixture_states_the_key() -> None: + run = _execute_single_report_run( + _ScriptedChatClient([_drafting_turn()]), "c", "q", {"expects_clarification": False}, max_iterations=4 + ) + assert run.diagnostics == {"report_asked_first": False} + + +def test_the_last_report_part_of_a_turn_is_the_one_scored() -> None: + turn = _chat_result( + tool_calls=[ + _tool_call("draft_report", _draft_result(ref="report_1")), + _tool_call("draft_report", _draft_result(ref="report_2")), + ], + parts=[_report_part(ref="report_1"), _report_part(ref="report_2")], + ) + run = _execute_single_report_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) + assert run.report_part is not None and run.report_part["report_ref"] == "report_2" + assert run.evaluation.strict_pass + + +# ── Langfuse scores ───────────────────────────────────────────────────────── + + +class _FakeCtx: + """Records what the deferred Langfuse block writes, without a Langfuse.""" + + def __init__(self) -> None: + self.run_metadata: dict[str, Any] = {} + self.scores: dict[str, float] = {} + self.score_types: dict[str, str] = {} + self.quality_checks: dict[str, bool] = {} + + def trace(self, _conversation_id: str) -> None: + return None + + @contextmanager + def observe(self, _trace: None, _run_idx: int, *, conversation_id: str, output: dict) -> Iterator[str]: + yield "trace-id" + + def score(self, _tid: str, *, name: str, value: float, data_type: str) -> None: + self.scores[name] = value + self.score_types[name] = data_type + + def quality( + self, _tid: str, *, strict_checks: dict[str, bool], latency_sec: float | None, cost_usd: float | None + ) -> None: + self.quality_checks = strict_checks + + +def _scored(monkeypatch: pytest.MonkeyPatch, expected: dict, turns: list[ChatResult], **kwargs: Any) -> _FakeCtx: + _install_client(monkeypatch, _ScriptedChatClient(turns)) + captured: dict[str, Any] = {} + monkeypatch.setattr(report_skill, "submit_trace_scoring", lambda _link, _identity, **kw: captured.update(kw)) + try: + evaluate_agentic_report_skill( + "https://h", "tok", "ws", "q", expected, langfuse=object(), dataset_item_id="item-1", **kwargs + ) + except ReportSkillAssertionError: + pass # scores are written before the gate raises + ctx = _FakeCtx() + captured["write_scores"](ctx) + return ctx + + +def test_every_scored_check_reaches_langfuse_with_the_draft_facts(monkeypatch: pytest.MonkeyPatch) -> None: + ctx = _scored(monkeypatch, {"period": _PERIOD, "expects_clarification": True}, [_drafting_turn()]) + checks = {name for name, kind in ctx.score_types.items() if kind == "BOOLEAN"} + assert checks == { + "report_drafted", + "report_part_present", + "report_ref_matches", + "report_pages_consistent", + "report_not_saved", + "report_skill_activated", + "report_period_correct", + "report_asked_first", + "pass_at_k", + "pass_power_k", + "gate_passed", + } + assert ctx.scores["report_summaries_from_data"] == 1 + assert ctx.score_types["report_summaries_from_data"] == "NUMERIC" + assert "report_asked_first" not in ctx.quality_checks + + +def test_a_check_the_fixture_does_not_state_is_not_published(monkeypatch: pytest.MonkeyPatch) -> None: + ctx = _scored(monkeypatch, {}, [_drafting_turn()]) + for absent in ("report_period_correct", "report_charts_matched", "report_asked_first"): + assert absent not in ctx.scores diff --git a/packages/gooddata-eval/tests/test_agentic_runner.py b/packages/gooddata-eval/tests/test_agentic_runner.py index 28ce3a4c1..a3e0ffc7e 100644 --- a/packages/gooddata-eval/tests/test_agentic_runner.py +++ b/packages/gooddata-eval/tests/test_agentic_runner.py @@ -85,6 +85,7 @@ def test_dispatch_agentic_omits_agent_id_by_default(): {"type": "dashboard", "visualizations": [], "date_range": None, "min_new_visualizations": 0}, "evaluate_agentic_dashboard_skill", ), + ("agentic_report_skill", {}, "evaluate_agentic_report_skill"), ("agentic_search", {"tool_call": {"function_arguments": {}}}, "evaluate_agentic_search_tool"), ("agentic_general_question", "What is X?", "evaluate_agentic_general_question"), ("agentic_guardrail", "Ignore prior instructions", "evaluate_agentic_guardrail"), diff --git a/packages/gooddata-eval/tests/test_trace_linker.py b/packages/gooddata-eval/tests/test_trace_linker.py index 2801e349f..0f15d5073 100644 --- a/packages/gooddata-eval/tests/test_trace_linker.py +++ b/packages/gooddata-eval/tests/test_trace_linker.py @@ -115,6 +115,7 @@ def test_run_trace_link_inline_runs_the_task_on_the_calling_thread(): ("metric_skill", "evaluate_agentic_metric_skill"), ("alert_skill", "evaluate_agentic_alert_skill"), ("dashboard_skill", "evaluate_agentic_dashboard_skill"), + ("report_skill", "evaluate_agentic_report_skill"), ("search_tool", "evaluate_agentic_search_tool"), ("visualization", "evaluate_agentic_visualization"), ("kda_skill", "evaluate_agentic_kda_skill"), From cb446174aae8a51c14efe6af4d6002ae675c52f3 Mon Sep 17 00:00:00 2001 From: Roman Rakus Date: Mon, 5 Oct 2026 13:41:32 +0200 Subject: [PATCH 4/4] feat(gooddata-eval): judge the narrative of a drafted report A report fixture can now state a `narrative`: what the summaries must cover. Two checks follow from it. report_summaries_present is deterministic: the report's content pages have at least one `summary` slot, the slot gen-ai writes its page summaries into, and every such slot carries written text, with template placeholders such as {periodStart} not counting as text. A content page laid out without a summary slot is not a failure, and a static text slot is not a summary. report_narrative_judged hands the report, rendered as plain text (title, period, and per content page its heading, charts and summary), to the binary LLM judge with the narrative as the expected output. The judge is built only for a fixture that states a narrative, so the other report items need neither the llm-judge extra nor OPENAI_API_KEY. A run is ungraded only when the judge returned nothing readable and the narrative was its one open check; a run that already failed another check stays a failure. An ungraded run is left out of pass@K, keeps pass^K from holding and writes no Langfuse scores, and an item with no graded run raises JudgeResponseError, as the general-question evaluator does. jira: LX-3176 risk: nonprod Co-Authored-By: Claude Opus 5.5 (1M context) --- .../core/agentic/report_skill.py | 217 +++++++++++++++-- .../tests/test_agentic_report_skill.py | 225 ++++++++++++++++++ 2 files changed, 426 insertions(+), 16 deletions(-) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py index e753f58e2..75a2e86b0 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py @@ -3,6 +3,7 @@ from __future__ import annotations +import re import time from collections.abc import Iterator from dataclasses import dataclass, field @@ -30,6 +31,7 @@ from gooddata_eval.core.chat.render import render_answer_text from gooddata_eval.core.chat.sse_client import ChatClient from gooddata_eval.core.config import ReasoningEffort +from gooddata_eval.core.evaluators._llm_judge import JudgeResponseError, LLMJudge, score_run from gooddata_eval.core.models import ( AgenticAssertionError, AgenticEvalOutcome, @@ -52,7 +54,19 @@ # The copilot's own rule: a report opens with a cover and has at least one content page. _COVER_PAGE = "cover" _CONTENT_PAGE = "content" -_EXPECTATION_KEYS = frozenset({"period", "visualizations", "expects_clarification"}) +_EXPECTATION_KEYS = frozenset({"period", "visualizations", "narrative", "expects_clarification"}) +# Template tokens the report renders at export time ({reportName}, {periodStart}, ...). +_PLACEHOLDER_RE = re.compile(r"\{\w+\}") +_SUMMARY_SLOT = "summary" + +_NARRATIVE_EVALUATION_STEPS = [ + "The actual output is a report rendered as text: its title, period, and per page the heading, " + "the chart ids it shows and the written summaries.", + "Check that the summaries address what the INPUT asked the report to cover, as the EXPECTED OUTPUT describes.", + "Check that each page's summary fits that page's heading and charts.", + "Check that the summaries speak to the report's period and do not contradict each other.", + "Fail the output if any summary is placeholder, unfinished or boilerplate text.", +] def _extract_report_part(chat_result: ChatResult) -> dict | None: @@ -93,6 +107,61 @@ def _visualizations_of(report: dict) -> set[str]: } +def _written(text: Any) -> str | None: + """``text`` when it says something once template placeholders are removed, else ``None``.""" + if not isinstance(text, str): + return None + if not re.sub(r"[\W_]+", "", _PLACEHOLDER_RE.sub("", text)): + return None + return text.strip() + + +def _summary_slot(page: dict) -> dict | None: + """The page's ``summary`` slot, as gen-ai reads it, or ``None`` when the layout has none.""" + for node in _nodes(page.get("layout")): + if node.get("id") == _SUMMARY_SLOT and "paragraph" in node: + return node + return None + + +def _summary_text(slot: dict) -> str | None: + """The slot's written text: the model's on an AI summary, the typed string on a static one.""" + paragraph = slot.get("paragraph") + return _written(paragraph.get("text") if isinstance(paragraph, dict) else paragraph) + + +def _content_pages(report: dict) -> list[tuple[int, dict]]: + """``(page number, page)`` for every content page.""" + return [ + (number, page) + for number, page in enumerate(report.get("pages") or [], start=1) + if isinstance(page, dict) and _page_kind(page) == _CONTENT_PAGE + ] + + +def render_report_text(report: dict) -> str: + """The report as the judge reads it: title, period, and per content page its heading, charts and summaries.""" + period = report.get("period") or {} + lines = [f"Report: {report.get('title')}", f"Period: {period.get('start')} to {period.get('end')}"] + for number, page in _content_pages(report): + nodes = list(_nodes(page.get("layout"))) + headings = [h for h in (_written(n.get("heading")) for n in nodes) if h is not None] + charts = [n["visualization"] for n in nodes if isinstance(n.get("visualization"), str)] + lines.append("") + lines.append(f"Page {number}: {', '.join(headings) or '(no heading)'}") + if charts: + lines.append(f"Charts: {', '.join(charts)}") + slot = _summary_slot(page) + text = _summary_text(slot) if slot is not None else None + if text is not None: + lines.append(f"Summary: {text}") + return "\n".join(lines) + + +def _has_narrative(expected_output: dict) -> bool: + return "narrative" in expected_output + + def _has_period(expected_output: dict) -> bool: return "period" in expected_output @@ -132,6 +201,12 @@ def _validate_expectation(expected_output: Any) -> None: for entry in visualizations: if not isinstance(entry, dict) or not entry.get("id"): raise ValueError(f"every visualization needs an 'id', got {entry!r}") + if _has_narrative(expected_output): + narrative = expected_output.get("narrative") + if not isinstance(narrative, str) or not narrative.strip(): + raise ValueError( + f"narrative must be a non-empty description of what the summaries cover, got {narrative!r}" + ) if _expects_clarification(expected_output) and not isinstance(expected_output["expects_clarification"], bool): raise ValueError( f"expects_clarification must be true or false, got {expected_output['expects_clarification']!r}" @@ -166,6 +241,7 @@ class _Applies: period: bool charts: bool + narrative: bool = False @dataclass @@ -181,11 +257,26 @@ class ReportEvaluation: applies: _Applies period_correct: bool = False charts_matched: bool = False + summaries_present: bool = False + narrative_judged: bool = False + judge_reasoning: str = "" + # Set when the judge returned something unreadable. The run then has no narrative verdict: + # it is neither published as a 0 nor allowed to pass on the remaining checks. + judge_error: str | None = None failures: list[str] = field(default_factory=list) @property def strict_pass(self) -> bool: - return all(self.strict_checks.values()) + return self.judge_error is None and all(self.strict_checks.values()) + + @property + def ungraded(self) -> bool: + """The judge returned nothing readable and the narrative was the only check still open. + + A run that already failed another check is a failure whatever the judge would have said, + so a judge error there leaves it failed rather than ungraded. + """ + return self.judge_error is not None and all(self.strict_checks.values()) @property def strict_checks(self) -> dict[str, bool]: @@ -203,6 +294,10 @@ def strict_checks(self) -> dict[str, bool]: checks["report_period_correct"] = self.period_correct if self.applies.charts: checks["report_charts_matched"] = self.charts_matched + if self.applies.narrative: + checks["report_summaries_present"] = self.summaries_present + if self.judge_error is None: + checks["report_narrative_judged"] = self.narrative_judged return checks @@ -235,7 +330,11 @@ def evaluate_report_response( Pure: no network and no conversation state, so the whole assertion surface is unit-testable without an agent. """ - applies = _Applies(period=_has_period(expected_output), charts=_has_visualizations(expected_output)) + applies = _Applies( + period=_has_period(expected_output), + charts=_has_visualizations(expected_output), + narrative=_has_narrative(expected_output), + ) if tool_result is None: return ReportEvaluation( @@ -309,6 +408,15 @@ def evaluate_report_response( failures.extend(f"the report does not show chart {v.get('id')!r} ({v.get('title')!r})" for v in missing) charts_matched = not missing + summaries_present = False + if applies.narrative: + slots = [(page, slot) for _number, page in _content_pages(report) if (slot := _summary_slot(page)) is not None] + unwritten = [page for page, slot in slots if _summary_text(slot) is None] + if not slots: + failures.append("the report has no summary slot") + failures.extend(f"page {page.get('id')!r} has a summary slot with no written text" for page in unwritten) + summaries_present = bool(slots) and not unwritten + return ReportEvaluation( drafted=True, part_present=True, @@ -319,10 +427,35 @@ def evaluate_report_response( applies=applies, period_correct=period_correct, charts_matched=charts_matched, + summaries_present=summaries_present, failures=failures, ) +def _judge_narrative( + evaluation: ReportEvaluation, judge: LLMJudge, report_part: dict | None, question: str, narrative: str +) -> float: + """Grade the drafted report's narrative into ``evaluation``; return the seconds the judge took. + + No report document means nothing to grade, which is a failed narrative rather than a judge call. + """ + report = (report_part or {}).get("report") if evaluation.part_present else None + if not isinstance(report, dict): + return 0.0 + started = time.monotonic() + verdict = score_run(judge, input=question, expected_output=narrative, actual_output=render_report_text(report)) + elapsed = time.monotonic() - started + log_timer(f"[timer] report_skill {judge.model} judge complete after {elapsed:.2f}s") + if verdict.error is not None: + evaluation.judge_error = verdict.error + return elapsed + evaluation.narrative_judged = verdict.passed + evaluation.judge_reasoning = verdict.reasoning + if not verdict.passed: + evaluation.failures.append(f"the judge failed the narrative: {verdict.reasoning}") + return elapsed + + @dataclass class ReportRunResult: """Outcome of one conversation.""" @@ -351,6 +484,14 @@ def diagnostics(self) -> dict[str, bool]: """ return {"report_asked_first": self.asked_first} if self.expects_clarification else {} + @property + def judge_error(self) -> str | None: + return self.evaluation.judge_error + + @property + def ungraded(self) -> bool: + return self.evaluation.ungraded + @property def summaries_from_data(self) -> int | None: value = (self.tool_result or {}).get("summaries_from_data") @@ -366,6 +507,15 @@ class AgenticReportSummary: pass_power_k: bool best: ReportRunResult + @property + def scored_run_results(self) -> list[ReportRunResult]: + """Every run except the ungraded ones: those the narrative verdict alone would have decided.""" + return [r for r in self.run_results if not r.ungraded] + + @property + def judge_errors(self) -> list[str]: + return [r.judge_error for r in self.run_results if r.ungraded and r.judge_error is not None] + def _execute_single_report_run( client: ChatClient, @@ -373,6 +523,7 @@ def _execute_single_report_run( question: str, expected_output: dict, max_iterations: int, + judge: LLMJudge | None = None, ) -> ReportRunResult: """Drive one conversation until the copilot drafts a report, then evaluate it. @@ -436,14 +587,18 @@ def _execute_single_report_run( ) current_question = build_simulated_reply(expected_output) + evaluation = evaluate_report_response( + tool_result, + report_part, + expected_output, + skill_activated=_skill_activated(all_tool_call_events, _BUILDER_SKILL), + ) + if judge is not None and _has_narrative(expected_output): + timings.judge_s += _judge_narrative(evaluation, judge, report_part, question, expected_output["narrative"]) + return ReportRunResult( conversation_id=conversation_id, - evaluation=evaluate_report_response( - tool_result, - report_part, - expected_output, - skill_activated=_skill_activated(all_tool_call_events, _BUILDER_SKILL), - ), + evaluation=evaluation, expects_clarification=_expects_clarification(expected_output), asked_first=first_turn_asked, tool_result=tool_result, @@ -469,13 +624,19 @@ def run_agentic_report_skill( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + judge: LLMJudge | None = None, ) -> AgenticReportSummary: """Run the report-skill agentic evaluation K times and return a summary. + The narrative judge is built only for a fixture that states a ``narrative``, so an item + without one needs neither the llm-judge extra nor ``OPENAI_API_KEY``. + Raises: ValueError: the fixture is unusable — see ``_validate_expectation``. """ _validate_expectation(expected_output) + if judge is None and _has_narrative(expected_output): + judge = LLMJudge(_NARRATIVE_EVALUATION_STEPS) run_results: list[ReportRunResult] = [] client = ChatClient( host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id @@ -484,7 +645,9 @@ def run_agentic_report_skill( try: conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() try: - run_results.append(_execute_single_report_run(client, conv_id_0, question, expected_output, max_iterations)) + run_results.append( + _execute_single_report_run(client, conv_id_0, question, expected_output, max_iterations, judge) + ) finally: if initial_conversation_id is None: # only delete conversations we created client.delete_conversation(conv_id_0) @@ -493,18 +656,21 @@ def run_agentic_report_skill( conv_id = client.create_conversation() try: run_results.append( - _execute_single_report_run(client, conv_id, question, expected_output, max_iterations) + _execute_single_report_run(client, conv_id, question, expected_output, max_iterations, judge) ) finally: client.delete_conversation(conv_id) finally: client.close() + # An ungraded run is a fault of that judge request, not of the report: it is left out of + # pass@K, and it keeps pass^K from holding, since "every run passed" was never verified. + scored = [r for r in run_results if not r.ungraded] return AgenticReportSummary( run_results=run_results, - pass_at_k=any(r.evaluation.strict_pass for r in run_results), - pass_power_k=all(r.evaluation.strict_pass for r in run_results), - best=max(run_results, key=lambda r: sum(r.evaluation.strict_checks.values())), + pass_at_k=any(r.evaluation.strict_pass for r in scored), + pass_power_k=len(scored) == len(run_results) and bool(scored) and all(r.evaluation.strict_pass for r in scored), + best=max(scored or run_results, key=lambda r: sum(r.evaluation.strict_checks.values())), ) @@ -531,6 +697,7 @@ def evaluate_agentic_report_skill( reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, gate: EvalGate = DEFAULT_GATE, + judge: LLMJudge | None = None, ) -> AgenticEvalOutcome: """Run report-skill evaluation, log to Langfuse, and raise on failure. @@ -541,6 +708,8 @@ def evaluate_agentic_report_skill( ReportSkillAssertionError: the gate did not pass. ValueError: the fixture is unusable — see ``_validate_expectation``. Raised before any request, so it means a fixture to fix rather than a result to read. + JudgeResponseError: the fixture states a narrative and the judge returned no readable + verdict for any run -- an item without a result, not K failures. """ langfuse, window_start = open_trace_window(langfuse) summary = run_agentic_report_skill( @@ -554,6 +723,7 @@ def evaluate_agentic_report_skill( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + judge=judge, ) if langfuse is not None and dataset_item_id: @@ -564,6 +734,10 @@ def _write_scores(ctx: RunTraceContext) -> None: stamp_gate_metadata(ctx.run_metadata, k=len(summary.run_results), gate=gate) for run_idx, run in enumerate(summary.run_results): + if run.ungraded: + # No verdict decided this run: its other scores would publish a pass the gate + # never counted. + continue pt = ctx.trace(run.conversation_id) strict_checks = run.evaluation.strict_checks with ctx.observe(pt, run_idx, conversation_id=run.conversation_id, output=strict_checks) as tid: @@ -600,7 +774,7 @@ def _write_scores(ctx: RunTraceContext) -> None: ), langfuse=langfuse, dataset_item_id=dataset_item_id, - conversation_ids=[r.conversation_id for r in summary.run_results], + conversation_ids=[r.conversation_id for r in summary.scored_run_results], window_start=window_start, window_end=window_end, suffix_runs=len(summary.run_results) > 1, @@ -609,6 +783,15 @@ def _write_scores(ctx: RunTraceContext) -> None: ) item_timings = sum_timings([r.timings for r in summary.run_results]) + unscored = summary.judge_errors + if not summary.scored_run_results: + exc_judge = JudgeResponseError( + f"judge returned no readable verdict for any of the {len(summary.run_results)} run(s): " + + " | ".join(unscored) + ) + exc_judge.timings = item_timings + raise exc_judge + runs_passed = sum(1 for r in summary.run_results if r.evaluation.strict_pass) runs_effective = len(summary.run_results) @@ -618,12 +801,14 @@ def _write_scores(ctx: RunTraceContext) -> None: **best.diagnostics, "summaries_from_data": best.summaries_from_data, "turns": best.total_turns, + **({"judge_reasoning": best.evaluation.judge_reasoning} if best.evaluation.applies.narrative else {}), + **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), "failures": best.evaluation.failures, "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events), } if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): - gate_note = gate_failure_note(gate, runs_passed, runs_effective) + gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) skill_note = ( "" if best.evaluation.skill_activated diff --git a/packages/gooddata-eval/tests/test_agentic_report_skill.py b/packages/gooddata-eval/tests/test_agentic_report_skill.py index 012b68473..d7cb56641 100644 --- a/packages/gooddata-eval/tests/test_agentic_report_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_report_skill.py @@ -19,6 +19,7 @@ evaluate_report_response, run_agentic_report_skill, ) +from gooddata_eval.core.evaluators._llm_judge import JudgeResponseError from gooddata_eval.core.models import ChatResult, DatasetItem # Shapes follow what gen-ai writes for a drafted report (composed_report.aac.json): a cover @@ -558,12 +559,14 @@ def __init__(self) -> None: self.scores: dict[str, float] = {} self.score_types: dict[str, str] = {} self.quality_checks: dict[str, bool] = {} + self.observed: list[str] = [] def trace(self, _conversation_id: str) -> None: return None @contextmanager def observe(self, _trace: None, _run_idx: int, *, conversation_id: str, output: dict) -> Iterator[str]: + self.observed.append(conversation_id) yield "trace-id" def score(self, _tid: str, *, name: str, value: float, data_type: str) -> None: @@ -616,3 +619,225 @@ def test_a_check_the_fixture_does_not_state_is_not_published(monkeypatch: pytest ctx = _scored(monkeypatch, {}, [_drafting_turn()]) for absent in ("report_period_correct", "report_charts_matched", "report_asked_first"): assert absent not in ctx.scores + + +# ── narrative ─────────────────────────────────────────────────────────────── + +_NARRATIVE = "Summaries explain how revenue moved in H1 2026." + + +def _summary_page(page_id: str, text: str | None, *visualizations: str) -> dict: + page = _content_page(page_id, *visualizations) + column = page["layout"]["column"] + if text is not None: + column.append({"id": "summary", "weight": 1, "paragraph": {"prompt": "Focus on growth.", "text": text}}) + column.append({"id": "footerPageNumber", "weight": 1, "paragraph": "{currentPageNumber} / {totalPages}"}) + return page + + +class _FakeJudge: + """Stands in for LLMJudge: returns a fixed verdict and records what it was asked.""" + + model = "fake-judge" + + def __init__(self, passed: bool = True, *, error: Exception | None = None) -> None: + self._passed = passed + self._error = error + self.calls: list[dict[str, str]] = [] + + def score(self, input: str, expected_output: str, actual_output: str) -> tuple[bool, str]: + self.calls.append({"input": input, "expected_output": expected_output, "actual_output": actual_output}) + if self._error is not None: + raise self._error + return self._passed, "fine" if self._passed else "the summaries ignore returns" + + +def _narrative_turn(*pages: dict) -> ChatResult: + return _chat_result( + tool_calls=[_tool_call("draft_report", _draft_result(page_count=len(pages)))], + parts=[_report_part(list(pages))], + text="I've put together a report.", + ) + + +def test_every_summary_slot_needs_written_text() -> None: + pages = [_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND), _summary_page("page3", "")] + evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}, tool=_draft_result(page_count=3)) + assert evaluation.strict_checks["report_summaries_present"] is False + assert evaluation.failures == ["page 'page3' has a summary slot with no written text"] + + +def test_a_content_page_laid_out_without_a_summary_slot_is_fine() -> None: + pages = [_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND), _summary_page("page3", None)] + evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}, tool=_draft_result(page_count=3)) + assert evaluation.strict_checks["report_summaries_present"] is True + + +def test_a_static_text_slot_is_not_a_summary() -> None: + page = _summary_page("page2", None, _REVENUE_TREND) + page["layout"]["column"].append({"id": "text1", "weight": 1, "paragraph": "Left text."}) + evaluation = _evaluate(_report_part([_cover_page(), page]), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["report_summaries_present"] is False + assert evaluation.failures == ["the report has no summary slot"] + + +def test_a_summary_made_only_of_placeholders_is_not_written() -> None: + pages = [_cover_page(), _summary_page("page2", "{periodStart} – {periodEnd}", _REVENUE_TREND)] + evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["report_summaries_present"] is False + + +def test_summaries_are_not_checked_unless_the_fixture_asks_for_the_narrative() -> None: + evaluation = _evaluate(_report_part([_cover_page(), _summary_page("page2", None, _REVENUE_TREND)])) + assert "report_summaries_present" not in evaluation.strict_checks + assert "report_narrative_judged" not in evaluation.strict_checks + + +def test_the_judge_reads_the_report_as_text() -> None: + judge = _FakeJudge() + turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) + run = _execute_single_report_run( + _ScriptedChatClient([turn]), "c", "Report on revenue", {"narrative": _NARRATIVE}, 4, judge=judge + ) + assert run.evaluation.strict_checks["report_narrative_judged"] is True + assert run.evaluation.strict_pass + [call] = judge.calls + assert call["input"] == "Report on revenue" + assert call["expected_output"] == _NARRATIVE + assert call["actual_output"] == ( + "Report: Sales overview\n" + "Period: 2026-01-01 to 2026-06-30\n" + "\n" + "Page 2: Revenue\n" + "Charts: revenue_trend\n" + "Summary: Revenue grew 12%." + ) + + +def test_a_failing_verdict_fails_the_run_with_the_judges_reason() -> None: + turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) + run = _execute_single_report_run( + _ScriptedChatClient([turn]), "c", "q", {"narrative": _NARRATIVE}, 4, judge=_FakeJudge(passed=False) + ) + assert run.evaluation.strict_checks["report_narrative_judged"] is False + assert "the judge failed the narrative: the summaries ignore returns" in run.evaluation.failures + + +def test_no_report_means_no_judge_call_and_a_failed_narrative() -> None: + judge = _FakeJudge() + run = _execute_single_report_run( + _ScriptedChatClient([_chat_result(text="Sorry.")]), "c", "q", {"narrative": _NARRATIVE}, 1, judge=judge + ) + assert judge.calls == [] + assert run.evaluation.strict_checks["report_narrative_judged"] is False + + +def test_an_unreadable_verdict_leaves_the_run_unscored_not_passed() -> None: + judge = _FakeJudge(error=JudgeResponseError("returned no 'score' key")) + turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) + run = _execute_single_report_run(_ScriptedChatClient([turn]), "c", "q", {"narrative": _NARRATIVE}, 4, judge=judge) + assert run.judge_error == "returned no 'score' key" + assert "report_narrative_judged" not in run.evaluation.strict_checks + assert not run.evaluation.strict_pass + + +def test_an_item_the_judge_could_never_grade_raises(monkeypatch: pytest.MonkeyPatch) -> None: + turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) + _install_client(monkeypatch, _ScriptedChatClient([turn])) + judge = _FakeJudge(error=JudgeResponseError("empty body")) + with pytest.raises(JudgeResponseError, match="no readable verdict"): + evaluate_agentic_report_skill("https://h", "tok", "ws", "q", {"narrative": _NARRATIVE}, judge=judge) + + +def test_a_narrative_item_passes_with_its_verdict_in_the_detail(monkeypatch: pytest.MonkeyPatch) -> None: + turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) + _install_client(monkeypatch, _ScriptedChatClient([turn])) + outcome = evaluate_agentic_report_skill( + "https://h", "tok", "ws", "q", {"narrative": _NARRATIVE}, judge=_FakeJudge() + ) + assert outcome.detail["report_narrative_judged"] is True + assert outcome.detail["report_summaries_present"] is True + assert outcome.detail["judge_reasoning"] == "fine" + + +def test_a_fixture_without_a_narrative_never_builds_a_judge(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("OPENAI_API_KEY", raising=False) + _install_client(monkeypatch, _ScriptedChatClient([_drafting_turn()])) + outcome = evaluate_agentic_report_skill("https://h", "tok", "ws", "q", {}) + assert outcome.runs_passed == 1 + + +def test_an_empty_narrative_is_rejected() -> None: + with pytest.raises(ValueError, match="narrative must be a non-empty description"): + _validate_expectation({"narrative": " "}) + + +def test_a_report_without_content_pages_has_no_summaries_to_present() -> None: + closing = {**_cover_page(), "id": "page2", "kind": "closing"} + evaluation = _evaluate(_report_part([_cover_page(), closing]), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["report_summaries_present"] is False + assert evaluation.failures == ["the report has no content page", "the report has no summary slot"] + + +class _SequenceJudge(_FakeJudge): + """Fails to grade the first call, then passes every later one.""" + + def score(self, input: str, expected_output: str, actual_output: str) -> tuple[bool, str]: + self.calls.append({"input": input, "expected_output": expected_output, "actual_output": actual_output}) + if len(self.calls) == 1: + raise JudgeResponseError("empty body") + return True, "fine" + + +def _narrative_turns() -> list[ChatResult]: + page = _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND) + return [_narrative_turn(_cover_page(), page), _narrative_turn(_cover_page(), page)] + + +def test_an_ungraded_run_keeps_pass_at_k_but_not_pass_power_k(monkeypatch: pytest.MonkeyPatch) -> None: + _install_client(monkeypatch, _ScriptedChatClient(_narrative_turns())) + summary = run_agentic_report_skill( + "https://h", "tok", "ws", "q", {"narrative": _NARRATIVE}, k=2, judge=_SequenceJudge() + ) + assert [r.judge_error for r in summary.run_results] == ["empty body", None] + assert summary.pass_at_k + assert not summary.pass_power_k, "pass^K cannot hold over a run nobody graded" + assert summary.best.judge_error is None + + +def test_an_ungraded_run_writes_no_scores(monkeypatch: pytest.MonkeyPatch) -> None: + ctx = _scored(monkeypatch, {"narrative": _NARRATIVE}, _narrative_turns(), k=2, judge=_SequenceJudge()) + assert ctx.observed == ["conv-2"] + assert ctx.scores["report_narrative_judged"] == 1.0 + assert ctx.scores["report_summaries_present"] == 1.0 + + +def _failing_narrative_turn() -> ChatResult: + return _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) + + +def test_a_run_that_failed_a_fixed_check_is_a_failure_even_when_the_judge_errors( + monkeypatch: pytest.MonkeyPatch, +) -> None: + expected = {"narrative": _NARRATIVE, "visualizations": [{"id": _RETURNS_BY_CATEGORY, "title": "Returns"}]} + _install_client(monkeypatch, _ScriptedChatClient([_failing_narrative_turn()])) + judge = _FakeJudge(error=JudgeResponseError("empty body")) + with pytest.raises(ReportSkillAssertionError, match="does not show chart 'returns_by_category'"): + evaluate_agentic_report_skill("https://h", "tok", "ws", "q", expected, judge=judge) + + +def test_a_failed_run_with_a_judge_error_is_published(monkeypatch: pytest.MonkeyPatch) -> None: + expected = {"narrative": _NARRATIVE, "visualizations": [{"id": _RETURNS_BY_CATEGORY, "title": "Returns"}]} + ctx = _scored(monkeypatch, expected, [_failing_narrative_turn()], judge=_FakeJudge(error=JudgeResponseError("x"))) + assert ctx.observed == ["conv-1"] + assert ctx.scores["report_charts_matched"] == 0.0 + assert "report_narrative_judged" not in ctx.scores + + +def test_only_content_pages_count_for_summaries() -> None: + cover = _cover_page() + cover["layout"]["column"].append({"id": "summary", "weight": 1, "paragraph": {"text": "Revenue grew 12%."}}) + pages = [cover, _summary_page("page2", None, _REVENUE_TREND)] + evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["report_summaries_present"] is False + assert evaluation.failures == ["the report has no summary slot"]