Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 16 additions & 1 deletion packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill
from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
from gooddata_eval.core.agentic.report_skill import evaluate_agentic_report_skill
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
from gooddata_eval.core.agentic.what_if import evaluate_agentic_what_if
Expand All @@ -43,6 +44,7 @@ class _LfKw(TypedDict, total=False):
"agentic_metric_skill",
"agentic_alert_skill",
"agentic_dashboard_skill",
"agentic_report_skill",
"agentic_search",
"agentic_general_question",
"agentic_guardrail",
Expand Down Expand Up @@ -92,7 +94,8 @@ class _LfKw(TypedDict, total=False):
#
# agentic_dashboard_skill is absent by default rather than by evidence: gen-ai holds the draft and
# any chart it authors in conversation state and writes neither until a user saves from the UI, so
# it is a candidate for the allowlist once the dataset has runs behind it.
# it is a candidate for the allowlist once the dataset has runs behind it. agentic_report_skill is
# absent for the same reason: the report draft stays in conversation state until a user saves it.
WORKSPACE_MUTATING_TEST_KINDS = frozenset(AGENTIC_TEST_KINDS) - PARALLEL_SAFE_TEST_KINDS


Expand Down Expand Up @@ -203,6 +206,18 @@ def _dispatch_agentic(
agent_id=agent_id,
**lf_kw,
)
elif kind == "agentic_report_skill":
return evaluate_agentic_report_skill(
host=host,
token=token,
workspace_id=workspace_id,
question=item.question,
expected_output=eo,
k=k,
gate=gate,
agent_id=agent_id,
**lf_kw,
)
elif kind == "agentic_alert_skill":
return evaluate_agentic_alert_skill(
host=host,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,14 @@
evaluate_agentic_metric_skill,
run_agentic_metric_skill,
)
from gooddata_eval.core.agentic.report_skill import (
AgenticReportSummary,
ReportEvaluation,
ReportRunResult,
ReportSkillAssertionError,
evaluate_agentic_report_skill,
run_agentic_report_skill,
)
from gooddata_eval.core.agentic.search_tool import (
AgenticSearchSummary,
SearchResult,
Expand All @@ -75,6 +83,7 @@
"AgenticGuardrailSummary",
"AgenticKdaSummary",
"AgenticMetricSummary",
"AgenticReportSummary",
"AgenticSearchSummary",
"AgenticRunSummary",
"AlertEvaluation",
Expand All @@ -95,6 +104,9 @@
"KdaSkillAssertionError",
"MetricRunResult",
"MetricSkillAssertionError",
"ReportEvaluation",
"ReportRunResult",
"ReportSkillAssertionError",
"RunResult",
"SearchResult",
"SearchToolAssertionError",
Expand All @@ -108,6 +120,7 @@
"evaluate_agentic_guardrail",
"evaluate_agentic_kda_skill",
"evaluate_agentic_metric_skill",
"evaluate_agentic_report_skill",
"evaluate_agentic_search_tool",
"evaluate_agentic_visualization",
"run_agentic_alert_skill",
Expand All @@ -117,6 +130,7 @@
"run_agentic_guardrail",
"run_agentic_kda_skill",
"run_agentic_metric_skill",
"run_agentic_report_skill",
"run_agentic_search_tool",
"run_agentic_visualization",
]
Loading
Loading