Skip to content

Commit ab98860

Browse files
Tomkessclaude
andcommitted
fix(gooddata-eval): apply the forecasting review findings to what-if
Two of the three findings on #1798 are structural and apply here unchanged. Tool calls were extracted from the current turn only. The agent may build the scenario spec on one turn and execute it on the next -- the common path, since it asks which measure to adjust first -- and reading a single turn dropped the scenario the execution actually ran, failing a correct run for having no adjustments. Extraction now reads every turn accumulated so far. Unasserted content checks were published to Langfuse as BOOLEAN 1. They are True internally so they cannot fail a run, but reporting that as a score claims the evaluator verified something it never looked at. Only checks named in ev.asserted are now scored. The third finding (unchecked confidence/seasonality) was forecasting-specific. 1 test added, verified to fail against the previous version. 803 passed, lint and format clean. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
1 parent 69f8538 commit ab98860

2 files changed

Lines changed: 37 additions & 6 deletions

File tree

‎packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py‎

Lines changed: 20 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -339,13 +339,16 @@ def _accumulate(result: ChatResult) -> None:
339339
reasoning_steps.extend(partial.reasoning_steps or [])
340340
response_id = partial.response_id or response_id
341341
_accumulate(partial)
342-
create_args, execute_result = _extract_what_if_calls(partial.tool_call_events or [])
342+
create_args, execute_result = _extract_what_if_calls(all_tool_call_events)
343343
turn_completed = False
344344
break
345345
reasoning_steps.extend(chat_result.reasoning_steps or [])
346346
response_id = chat_result.response_id or response_id
347347
_accumulate(chat_result)
348-
create_args, execute_result = _extract_what_if_calls(chat_result.tool_call_events or [])
348+
# Over every turn so far, not just this one: the agent may build the spec on
349+
# one turn and execute it on the next, and reading a single turn would drop the
350+
# scenario the execution actually ran.
351+
create_args, execute_result = _extract_what_if_calls(all_tool_call_events)
349352
response_text = render_answer_text(chat_result)
350353
turn_completed = chat_result.stream_ended and bool(response_text)
351354
if execute_result is not None:
@@ -493,11 +496,22 @@ def _write_scores(ctx: RunTraceContext) -> None:
493496
"what_if_executed": ev.executed,
494497
"what_if_success": ev.success,
495498
"what_if_turn_completed": ev.turn_completed,
496-
"what_if_metric_correct": ev.metric_correct,
497-
"what_if_maql_correct": ev.maql_correct,
498-
"what_if_scenario_count_correct": ev.scenario_count_correct,
499-
"what_if_baseline_correct": ev.baseline_correct,
500499
}
500+
# Only the content checks the fixture actually pinned. An unasserted check
501+
# is True internally so it cannot fail a run, but publishing that as a
502+
# BOOLEAN 1 would claim the evaluator verified something it never looked at.
503+
strict_checks.update(
504+
{
505+
key: value
506+
for name, key, value in (
507+
("metric_id", "what_if_metric_correct", ev.metric_correct),
508+
("scenario_maql", "what_if_maql_correct", ev.maql_correct),
509+
("scenarios", "what_if_scenario_count_correct", ev.scenario_count_correct),
510+
("include_baseline", "what_if_baseline_correct", ev.baseline_correct),
511+
)
512+
if name in ev.asserted
513+
}
514+
)
501515
with ctx.observe(pt, run_idx) as tid:
502516
for score_name, value in strict_checks.items():
503517
ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN")

‎packages/gooddata-eval/tests/test_agentic_what_if.py‎

Lines changed: 17 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -285,3 +285,20 @@ def test_a_wrong_adjustment_raises_naming_what_the_agent_actually_did():
285285
assert "* 3 *" in str(error)
286286
assert error.runs_passed == 0
287287
assert error.conversation_id == "conv-1"
288+
289+
290+
def test_a_spec_built_on_an_earlier_turn_is_still_the_one_scored():
291+
"""The agent may build the spec on one turn and execute it on the next -- reading only
292+
the current turn's calls would drop the scenario the execution actually ran and fail a
293+
correct run for having no adjustments."""
294+
summary = _run(
295+
[
296+
_chat([_tc("create_what_if_scenario", _create_args())], text="Preparing the scenario."),
297+
_chat([_tc("execute_what_if_scenario", {"scenario_ref": "wia_1"}, _OK_EXECUTE)]),
298+
]
299+
)
300+
301+
assert summary.best.evaluation.executed is True
302+
assert summary.best.evaluation.metric_correct is True
303+
assert summary.best.evaluation.maql_correct is True
304+
assert summary.pass_at_k is True

0 commit comments

Comments
 (0)