Skip to content

Commit fc15e03

Browse files
FrankHuynhclaude
andcommitted
feat(gooddata-eval): evaluate the agentic dashboard-creation skill
Adds the agentic_dashboard_skill kind: it drives the conversation until the agent produces a draft, then scores that draft. The gate holds only what the agent decided -- the draft call succeeded, the response carries the expected part, every expected chart is present, every tab's date filter matches, and it authored at least the required number of charts. Two observations sit beside the gate and never fail a run: whether a set_skills call activated dashboard_builder, and whether the response's references carried every widget id. The second one matters -- gen-ai rejects a draft naming an unresolvable visualization before it can succeed, so a widget missing from the references means reference building degraded on the way out, which is a platform fault and must not be scored as the model's. A chart the fixture gives an id is matched on that id alone, since the widget title is title_override or fallback_title and the override is the model's own choice; a differing title is reported as a note. A chart the fixture marks with a null id has no identity to match on, so there the title is the match, taken among the authored charts only -- matching it against every chart would let an existing one of the same title satisfy it. The simulated user is deterministic rather than an LLM call: when the agent asks back instead of drafting, the only things it still needs are the charts and the date range, and both come straight from the expectation. That reply is byte-identical every turn, which is why the loop caps at one clarifying round instead of inheriting metric_skill's seven. The expectation is validated before the first API call: a fixture with no visualizations would pass every chart check vacuously, and a date range the simulated user cannot phrase would otherwise surface only on the branch where the agent asks back, passing or crashing depending on what the model chose. jira: QA-29347 risk: low Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
1 parent 1ab844c commit fc15e03

6 files changed

Lines changed: 1278 additions & 0 deletions

File tree

‎packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -13,6 +13,7 @@
1313
from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
1414
from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
1515
from gooddata_eval.core.agentic.conversation import ConversationFixture, evaluate_agentic_conversation
16+
from gooddata_eval.core.agentic.dashboard_skill import evaluate_agentic_dashboard_skill
1617
from gooddata_eval.core.agentic.general_question import evaluate_agentic_general_question
1718
from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
1819
from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill
@@ -40,6 +41,7 @@ class _LfKw(TypedDict, total=False):
4041
"agentic_visualization", # experimental: expected_output.expected_outputs (multi-candidate)
4142
"agentic_metric_skill",
4243
"agentic_alert_skill",
44+
"agentic_dashboard_skill",
4345
"agentic_search",
4446
"agentic_general_question",
4547
"agentic_guardrail",
@@ -85,6 +87,10 @@ class _LfKw(TypedDict, total=False):
8587
# create_key_driver_analysis with no cleanup, and while the evaluator only ever reads that
8688
# call's ARGUMENTS -- never a created object id -- whether the platform persists anything is
8789
# unverified. Move it to the allowlist once someone confirms it does not.
90+
#
91+
# agentic_dashboard_skill is absent by default rather than by evidence: gen-ai holds the draft and
92+
# any chart it authors in conversation state and writes neither until a user saves from the UI, so
93+
# it is a candidate for the allowlist once the dataset has runs behind it.
8894
WORKSPACE_MUTATING_TEST_KINDS = frozenset(AGENTIC_TEST_KINDS) - PARALLEL_SAFE_TEST_KINDS
8995

9096

@@ -183,6 +189,18 @@ def _dispatch_agentic(
183189
agent_id=agent_id,
184190
**lf_kw,
185191
)
192+
elif kind == "agentic_dashboard_skill":
193+
return evaluate_agentic_dashboard_skill(
194+
host=host,
195+
token=token,
196+
workspace_id=workspace_id,
197+
question=item.question,
198+
expected_output=eo if isinstance(eo, dict) else {},
199+
k=k,
200+
gate=gate,
201+
agent_id=agent_id,
202+
**lf_kw,
203+
)
186204
elif kind == "agentic_alert_skill":
187205
return evaluate_agentic_alert_skill(
188206
host=host,

‎packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -16,6 +16,16 @@
1616
evaluate_agentic_conversation,
1717
run_agentic_conversation,
1818
)
19+
from gooddata_eval.core.agentic.dashboard_skill import (
20+
AgenticDashboardSummary,
21+
DashboardEvaluation,
22+
DashboardRunResult,
23+
DashboardSkillAssertionError,
24+
build_simulated_reply,
25+
evaluate_agentic_dashboard_skill,
26+
evaluate_dashboard_draft,
27+
run_agentic_dashboard_skill,
28+
)
1929
from gooddata_eval.core.agentic.general_question import (
2030
AgenticGeneralQuestionSummary,
2131
GeneralQuestionAssertionError,
@@ -62,6 +72,7 @@
6272

6373
__all__ = [
6474
"AgenticAlertSummary",
75+
"AgenticDashboardSummary",
6576
"AgenticGeneralQuestionSummary",
6677
"AgenticGuardrailSummary",
6778
"AgenticKdaSummary",
@@ -74,6 +85,9 @@
7485
"ConversationAssertionError",
7586
"ConversationFixture",
7687
"ConversationResult",
88+
"DashboardEvaluation",
89+
"DashboardRunResult",
90+
"DashboardSkillAssertionError",
7791
"GeneralQuestionAssertionError",
7892
"GeneralQuestionResult",
7993
"GuardrailAssertionError",
@@ -91,6 +105,9 @@
91105
"VisualizationAssertionError",
92106
"evaluate_agentic_alert_skill",
93107
"evaluate_agentic_conversation",
108+
"build_simulated_reply",
109+
"evaluate_agentic_dashboard_skill",
110+
"evaluate_dashboard_draft",
94111
"evaluate_agentic_general_question",
95112
"evaluate_agentic_guardrail",
96113
"evaluate_agentic_kda_skill",
@@ -99,6 +116,7 @@
99116
"evaluate_agentic_visualization",
100117
"run_agentic_alert_skill",
101118
"run_agentic_conversation",
119+
"run_agentic_dashboard_skill",
102120
"run_agentic_general_question",
103121
"run_agentic_guardrail",
104122
"run_agentic_kda_skill",

0 commit comments

Comments
 (0)