Skip to content

Commit 782acee

Browse files
Tomkessclaude
andauthored
chore(gooddata-eval): use neutral identifiers in fixtures, drop a hardcoded path (#1827)
Test fixtures carried metric and label names copied out of one particular workspace, and two comments named an external repository rather than saying what they meant. Neither tells a reader of this package anything: the scoring tests exercise attribute-filter ordering and the names are arbitrary, so they now use generic ones. verify_guardrail_refusal_criteria.py loaded .env from an absolute path under one developer's home directory, which made the script runnable on exactly one machine. It now reads GD_EVAL_ENV_FILE, defaulting to .env in the working directory. One test needed care rather than a rename: the two sides of test_the_same_elements_on_a_different_label_still_differ deliberately name different labels, and renaming both to the same thing quietly turned it into a test of nothing. They stay distinct. No behaviour change. 1111 tests pass. Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
1 parent 15ae389 commit 782acee

4 files changed

Lines changed: 16 additions & 14 deletions

File tree

‎packages/gooddata-eval/scripts/verify_guardrail_refusal_criteria.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -39,7 +39,9 @@
3939

4040
from dotenv import load_dotenv
4141

42-
load_dotenv("/Users/petertomko/gdc-mic-ai-evaluation/.env")
42+
# Whatever .env the caller points at, defaulting to the working directory. It used to be an
43+
# absolute path, which made the script runnable on exactly one machine.
44+
load_dotenv(os.environ.get("GD_EVAL_ENV_FILE", ".env"))
4345

4446
from gooddata_eval.core.agentic.guardrail import _GUARDRAIL_EVALUATION_STEPS # noqa: E402
4547
from gooddata_eval.core.evaluators._guardrail_criteria import ( # noqa: E402

‎packages/gooddata-eval/src/gooddata_eval/core/models.py‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -134,8 +134,8 @@ class ReasoningStepEvent(BaseModel):
134134

135135
# Reasoning summaries are full paragraphs, e.g. "**Identifying analytics needs**\n\nI'm
136136
# analyzing..." -- using the whole thing as a latency_breakdown label would make every
137-
# entry an unreadable wall of text. Same bolded-title convention this repo's own reasoning
138-
# tooling already keys off of (see gdc-mic-ai-evaluation's generate_dashboard_summary.py).
137+
# entry an unreadable wall of text. The bolded title is the summary's own heading, and
138+
# downstream reporting keys off it for the same reason.
139139
_REASONING_TITLE_RE = re.compile(r"^\*\*(.+?)\*\*")
140140
_REASONING_LABEL_MAX_LEN = 60
141141

‎packages/gooddata-eval/tests/test_from_insights.py‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -382,15 +382,15 @@ def test_built_envelope_is_loadable_as_a_dataset_item(tmp_path):
382382
],
383383
),
384384
)
385-
envelope = build(spec, "How did spend trend by month?", "micai_diagnose_master", set())
385+
envelope = build(spec, "How did spend trend by month?", "demo_workspace", set())
386386
assert "_shape" not in envelope["expected_output"]["visualization"]
387387
assert _validation_errors(envelope) is None
388388

389389
(tmp_path / f"{envelope['id']}.json").write_text(json.dumps(envelope, indent=2))
390390
items = load_local_dataset(tmp_path)
391391
assert [i.id for i in items] == [envelope["id"]]
392392
assert items[0].test_kind == "visualization"
393-
assert items[0].dataset_name == "micai_diagnose_master"
393+
assert items[0].dataset_name == "demo_workspace"
394394

395395

396396
def test_mint_id_is_stable_and_collision_safe():

‎packages/gooddata-eval/tests/test_scoring.py‎

Lines changed: 9 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -79,10 +79,10 @@ def test_check_filters_exact_attribute_match():
7979
# state, type) and never the LIST under state["include"], so two filters selecting the
8080
# same elements in a different order compare unequal.
8181
#
82-
# Found from a real eval run (gdc-mic-ai-evaluation, micai_diagnose_master, 2026-09-10):
83-
# a question filtering cross-border traffic scored metrics_correct=True,
84-
# dimensions_correct=True, filters_correct=False, because the fixture listed
85-
# ["Inter-region", "Intra-region"] and the agent emitted ["Intra-region", "Inter-region"].
82+
# Found from a real eval run: a question filtering on a two-value attribute scored
83+
# metrics_correct=True, dimensions_correct=True, filters_correct=False, because the fixture
84+
# listed ["Inter-region", "Intra-region"] and the agent emitted the same two the other way
85+
# round.
8686
# Element order is not something an agent has any reason to keep stable between runs, so
8787
# every question needing a multi-value attribute filter passes or fails partly at random.
8888

@@ -95,7 +95,7 @@ def viz(values):
9595
"filter_by": {
9696
"f_a": {
9797
"type": "attribute_filter",
98-
"using": "label/cross_border_name",
98+
"using": "label/region_name",
9999
"state": {"include": values},
100100
}
101101
},
@@ -286,10 +286,10 @@ def test_normalized_filters_is_empty_per_category_when_unfiltered():
286286
assert normalized_filters(viz) == {"date": [], "ranking": [], "attribute": []}
287287

288288

289-
def _attr_viz(values, key="include", using="label/cross_border_name"):
289+
def _attr_viz(values, key="include", using="label/region_name"):
290290
return _viz(
291291
query={
292-
"fields": {"m": {"using": "metric/approval_rate"}},
292+
"fields": {"m": {"using": "metric/conversion_rate"}},
293293
"filter_by": {"f": {"type": "attribute_filter", "using": using, "state": {key: values}}},
294294
},
295295
metrics=["m"],
@@ -337,8 +337,8 @@ def test_include_and_exclude_of_the_same_elements_still_differ():
337337

338338

339339
def test_the_same_elements_on_a_different_label_still_differ():
340-
expected = _attr_viz(["A", "B"], using="label/cross_border_name")
341-
assert check_filters(expected, _attr_viz(["B", "A"], using="label/region_name")).attribute_ok is False
340+
expected = _attr_viz(["A", "B"], using="label/region_name")
341+
assert check_filters(expected, _attr_viz(["B", "A"], using="label/channel_name")).attribute_ok is False
342342

343343

344344
def test_a_mixed_type_element_list_does_not_crash_scoring():

0 commit comments

Comments
 (0)