Skip to content

Commit 36ed3d0

Browse files
Tomkessclaude
andcommitted
fix(gooddata-eval): tell the simulated user the forecast settings it pins
`forecast_confidence` and `forecast_seasonal` are scored whenever a fixture pins them, but `_build_clarification_prompt` never passed them to the simulated user. An agent that asks which confidence level to use therefore got a guess, chose something else, and `confidence_correct` reported the harness's own omission as the agent's error. Both are now included on the same terms as the other hints, with `is not None` rather than a truth test: a pinned `forecast_seasonal: false` is an answer, not an absence, and `0.0` would be a confidence level. Also drops a `visualization_ref` argument from one extraction test. The production code never read it and no other test sends it -- it was decoration that implied an argument the execute_forecast payload does not carry, and reading it as real is what the pairing logic would have to do to be wrong. Found in review by CodeRabbit. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
1 parent 0cf23ca commit 36ed3d0

2 files changed

Lines changed: 51 additions & 1 deletion

File tree

‎packages/gooddata-eval/src/gooddata_eval/core/agentic/forecasting.py‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -77,6 +77,17 @@ def _build_clarification_prompt(agent_message: str, expected_output: dict) -> st
7777
granularity = expected_output.get("granularity")
7878
if granularity:
7979
hints.append(f"the time granularity is {granularity}")
80+
# Both are scored when the fixture pins them, so withholding them here would let the
81+
# harness fail a run on a value it refused to supply: the agent asks which confidence
82+
# level to use, the simulated user guesses, and `confidence_correct` reports the guess
83+
# as the agent's error. `is not None` rather than a truth test -- a pinned
84+
# `forecast_seasonal: false` is an answer, and `0.0` would be a confidence level.
85+
confidence = expected_output.get("forecast_confidence")
86+
if confidence is not None:
87+
hints.append(f"the confidence level is {confidence}")
88+
seasonal = expected_output.get("forecast_seasonal")
89+
if seasonal is not None:
90+
hints.append(f"seasonality {'should' if seasonal else 'should not'} be modelled")
8091
reference = "; ".join(hints)
8192
return (
8293
f"You are simulating a user in a conversation with a BI assistant that forecasts metric "

‎packages/gooddata-eval/tests/test_agentic_forecasting.py‎

Lines changed: 40 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,6 +6,7 @@
66
import pytest
77
from gooddata_eval.core.agentic.forecasting import (
88
ForecastingAssertionError,
9+
_build_clarification_prompt,
910
_evaluate_run,
1011
_extract_forecast_calls,
1112
_metric_uris,
@@ -76,7 +77,7 @@ def test_extract_pairs_the_execute_with_the_visualization_it_followed():
7677
would pair a fresh visualization with a stale result."""
7778
calls = [
7879
_tc("create_adhoc_visualization", _viz_args(period=1)),
79-
_tc("execute_forecast", {"visualization_ref": "viz_1"}, _OK_FORECAST),
80+
_tc("execute_forecast", {}, _OK_FORECAST),
8081
_tc("create_adhoc_visualization", _viz_args(period=3)),
8182
]
8283
viz, result = _extract_forecast_calls(calls)
@@ -487,3 +488,41 @@ def test_any_gate_still_passes_the_same_item():
487488
def test_the_default_gate_is_pass_at_k():
488489
outcome = _two_runs()
489490
assert outcome.runs_passed == 1
491+
492+
493+
# ── the simulated user's reply ──────────────────────────────────────────────
494+
495+
496+
def test_the_reply_carries_every_hint_the_fixture_pins():
497+
"""Anything the fixture pins is also scored, so withholding it would let the harness fail
498+
a run on a value it refused to supply: the agent asks which confidence level to use, the
499+
simulated user guesses, and `confidence_correct` reports the guess as the agent's error."""
500+
prompt = _build_clarification_prompt(
501+
"Which confidence level, and should I model seasonality?",
502+
{
503+
"metric": "metric/spend",
504+
"forecast_period": 3,
505+
"granularity": "MONTH",
506+
"forecast_confidence": 0.8,
507+
"forecast_seasonal": True,
508+
},
509+
)
510+
assert "metric/spend" in prompt
511+
assert "3 periods ahead" in prompt
512+
assert "MONTH" in prompt
513+
assert "confidence level is 0.8" in prompt
514+
assert "seasonality should be modelled" in prompt
515+
516+
517+
def test_a_pinned_false_seasonal_is_an_answer_not_an_absence():
518+
"""`is not None`, not a truth test. `forecast_seasonal: false` is scored -- the tool
519+
defaults it to false -- so the simulated user has to be able to say so."""
520+
prompt = _build_clarification_prompt("Seasonal?", {**_EXPECTED, "forecast_seasonal": False})
521+
assert "seasonality should not be modelled" in prompt
522+
523+
524+
def test_an_unpinned_hint_is_dropped_rather_than_asserted_as_none():
525+
prompt = _build_clarification_prompt("Which measure?", {"metric": "metric/spend"})
526+
assert "confidence" not in prompt
527+
assert "seasonality" not in prompt
528+
assert "None" not in prompt

0 commit comments

Comments
 (0)