Skip to content

Commit f182836

Browse files
Tomkessclaude
andcommitted
Merge branch 'master' into feat/agentic-dashboard-summary
#1799 landed on master, so the two evaluators now register side by side. The registry set, the parametrized kind list and the trace-linker list each gain both entries. The dispatch needed care rather than concatenation: the two elif branches share the k/agent_id/**lf_kw tail that follows the conflict marker, so keeping both headers alone would have spliced one argument tail onto two calls and dropped dashboard_summary's own arguments. Each kind now has its own complete call. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2 parents eaeb3d0 + 9be051b commit f182836

9 files changed

Lines changed: 1063 additions & 14 deletions

File tree

‎packages/gooddata-eval/scripts/verify_guardrail_refusal_criteria.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -39,7 +39,9 @@
3939

4040
from dotenv import load_dotenv
4141

42-
load_dotenv("/Users/petertomko/gdc-mic-ai-evaluation/.env")
42+
# Whatever .env the caller points at, defaulting to the working directory. It used to be an
43+
# absolute path, which made the script runnable on exactly one machine.
44+
load_dotenv(os.environ.get("GD_EVAL_ENV_FILE", ".env"))
4345

4446
from gooddata_eval.core.agentic.guardrail import _GUARDRAIL_EVALUATION_STEPS # noqa: E402
4547
from gooddata_eval.core.evaluators._guardrail_criteria import ( # noqa: E402

‎packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py‎

Lines changed: 13 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,7 @@
2121
from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
2222
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
2323
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
24+
from gooddata_eval.core.agentic.what_if import evaluate_agentic_what_if
2425
from gooddata_eval.core.config import ReasoningEffort
2526
from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, DatasetItem
2627
from gooddata_eval.core.runner import EvalReport, ItemReport
@@ -49,6 +50,7 @@ class _LfKw(TypedDict, total=False):
4950
"agentic_conversation",
5051
"agentic_kda_skill",
5152
"agentic_dashboard_summary",
53+
"agentic_what_if",
5254
}
5355
)
5456

@@ -288,6 +290,17 @@ def _dispatch_agentic(
288290
agent_id=agent_id,
289291
**lf_kw,
290292
)
293+
elif kind == "agentic_what_if":
294+
return evaluate_agentic_what_if(
295+
host=host,
296+
token=token,
297+
workspace_id=workspace_id,
298+
question=item.question,
299+
expected_output=eo if isinstance(eo, dict) else {},
300+
k=k,
301+
agent_id=agent_id,
302+
**lf_kw,
303+
)
291304
elif kind == "agentic_conversation":
292305
fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
293306
return evaluate_agentic_conversation(

0 commit comments

Comments
 (0)