From 8d8a6fed15d5e72ac1c74c754c2e78dd1e9551b0 Mon Sep 17 00:00:00 2001 From: "my.nguyen" Date: Tue, 6 Oct 2026 09:22:23 +0700 Subject: [PATCH] feat(gooddata-eval): relay user_context on every agentic kind An item's user_context reached the chat request only for agentic_general_question and the single-shot path; every other agentic kind asked its question bare, so a dataset item scoped to a dashboard or widget could not be evaluated anywhere else. ChatClient now takes a user_context that it sends as userContext on every message, and a per-call value still overrides it. Each run_agentic_* binds the item's context to its client rather than to the first message: gen-ai treats a message without a context as cleared, so the follow-up and clarification turns need it too. The dispatch forwards item.user_context to every kind, and each evaluate_agentic_* passes it to its runner, keeping gate last. agentic_conversation takes the context per turn. The item's context applies from the first turn; a turn that sets user_context replaces it from that turn on, null clears it, and a turn that omits the field keeps the current one, as the attached chip does in the UI. The simulated user's replies carry the turn's context, and a turn skipped for an unresolvable expectation still changes it. Tests cover the forwarding for every kind at each layer (dispatch, evaluator, client binding), with a check that the dispatch list matches AGENTIC_TEST_KINDS, plus the per-turn conversation semantics and the client default on the wire. The Langfuse loader now fails the dataset load on a user_context that is not an object, instead of skipping it: a skipped context asks the item bare, which fails for a reason unrelated to the item. The README documents user_context in the dataset format, including the per-turn semantics of agentic_conversation and the enableAiContextSetup flag. jira: QA-29627 risk: low --- packages/gooddata-eval/README.md | 49 +++++++++ .../src/gooddata_eval/cli/agentic_runner.py | 11 +++ .../gooddata_eval/core/agentic/alert_skill.py | 10 +- .../core/agentic/anomaly_detection.py | 10 +- .../core/agentic/conversation.py | 33 ++++++- .../core/agentic/dashboard_skill.py | 10 +- .../core/agentic/general_question.py | 18 ++-- .../gooddata_eval/core/agentic/guardrail.py | 10 +- .../gooddata_eval/core/agentic/kda_skill.py | 10 +- .../core/agentic/metric_skill.py | 10 +- .../core/agentic/report_skill.py | 10 +- .../gooddata_eval/core/agentic/search_tool.py | 10 +- .../core/agentic/visualization.py | 10 +- .../src/gooddata_eval/core/agentic/what_if.py | 10 +- .../src/gooddata_eval/core/chat/sse_client.py | 20 +++- .../core/dataset/langfuse_source.py | 15 ++- .../tests/test_agentic_conversation.py | 95 +++++++++++++++++- .../tests/test_agentic_general_question.py | 18 ---- .../tests/test_agentic_runner.py | 99 +++++++++++++++++-- .../tests/test_langfuse_source.py | 13 +++ .../gooddata-eval/tests/test_sse_client.py | 23 +++++ 21 files changed, 437 insertions(+), 57 deletions(-) diff --git a/packages/gooddata-eval/README.md b/packages/gooddata-eval/README.md index ac0b2c72e..d6b614955 100644 --- a/packages/gooddata-eval/README.md +++ b/packages/gooddata-eval/README.md @@ -574,6 +574,55 @@ The `expected_output` rubric: Each criterion is scored independently by the LLM judge, so `quality_score` is the fraction of satisfied criteria. +### Items asked in a dashboard context (`user_context`) + +An item can be asked the way a user asks from an open dashboard or an attached widget. +Its `user_context` is sent verbatim as `userContext` on every chat message of the item, +follow-up and clarification messages included — the server treats a message without one +as a cleared context. Requires the `enableAiContextSetup` feature flag on the target +organization — without it the server ignores the dashboard view. Applies to chat items +only; `dashboard_summary` items carry their scope in `summary_input` instead. No +`userContext` key is sent at all when neither the item nor, for `agentic_conversation`, +the current turn sets one. + +```json +{ + "id": "ctx-001", + "dataset_name": "my_dataset_ctx", + "test_kind": "general_question", + "question": "Which dashboard am I looking at, and what does it cover?", + "user_context": { + "view": { + "dashboard": { + "id": "sales_overview", + "title": "Sales Overview", + "widgets": [ + {"widgetType": "insight", "widgetId": "w1", "title": "Revenue by Month", "visualizationId": "revenue_by_month"} + ] + } + } + }, + "expected_output": "Names the Sales Overview dashboard and summarizes its charts." +} +``` + +The schema belongs to the AI chat API (`UserContext`: `view.dashboard`, +`referencedObjects`, `activeObject`), so it is not validated here beyond being an object. +In a Langfuse dataset, put it in the item `metadata` (or in the `input` object) under +`user_context`. An object in `input` wins and `metadata` is then not read; a missing or +`null` value in `input` falls back to `metadata`. The value used must be an object or `null`; +anything else fails the dataset load. + +`agentic_conversation` items take the item's `user_context` as the context of the first +turn. A turn in `expected_output.turns` can set its own `user_context`, which applies +from that turn on until another turn changes it, as the attached context does in the UI: + +| Turn | Context sent | +|---|---| +| no `user_context` key | the current one, unchanged | +| `"user_context": {...}` | this one, from this turn on | +| `"user_context": null` | none, from this turn on | + ## Supported test kinds | test_kind | What the agent must produce | Extra required | diff --git a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py index 60b7296f6..1f17dd5f8 100644 --- a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py @@ -182,6 +182,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_metric_skill": @@ -194,6 +195,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_dashboard_skill": @@ -206,6 +208,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_report_skill": @@ -218,6 +221,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_alert_skill": @@ -230,6 +234,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_search": @@ -245,6 +250,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_general_question": @@ -270,6 +276,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_kda_skill": @@ -282,6 +289,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_what_if": @@ -293,6 +301,7 @@ def _dispatch_agentic( expected_output=eo if isinstance(eo, dict) else {}, k=k, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_anomaly_detection": @@ -305,6 +314,7 @@ def _dispatch_agentic( k=k, gate=gate, agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) elif kind == "agentic_conversation": @@ -315,6 +325,7 @@ def _dispatch_agentic( workspace_id=workspace_id, fixture=ConversationFixture.model_validate(fixture_data), agent_id=agent_id, + user_context=item.user_context, **lf_kw, ) else: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py index c1228e89b..af9a7090b 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/alert_skill.py @@ -659,12 +659,18 @@ def run_agentic_alert_skill( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticAlertSummary: """Run the alert-skill agentic evaluation K times and return a summary.""" expected = _normalize_expected_output(expected_output) run_results: list[AlertRunResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) sdk = GoodDataSdk.create(host, token) @@ -858,6 +864,7 @@ def evaluate_agentic_alert_skill( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, gate: EvalGate = DEFAULT_GATE, ) -> AgenticEvalOutcome: """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure. @@ -881,6 +888,7 @@ def evaluate_agentic_alert_skill( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py index 778da4fea..97cb215fc 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/anomaly_detection.py @@ -344,6 +344,7 @@ def run_agentic_anomaly_detection( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticAnomalySummary: """Run the anomaly-detection agentic evaluation K times and return a summary. @@ -356,7 +357,12 @@ def run_agentic_anomaly_detection( raise ValueError(f"k must be >= 1, got {k}") run_results: list[AnomalyRunResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) def _run_once(conv_id: str) -> AnomalyRunResult: @@ -520,6 +526,7 @@ def evaluate_agentic_anomaly_detection( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, ) -> AgenticEvalOutcome: """Run anomaly-detection evaluation, log to Langfuse, and raise on failure.""" langfuse, window_start = open_trace_window(langfuse) @@ -534,6 +541,7 @@ def evaluate_agentic_anomaly_detection( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py index 695295a7b..25e0260d0 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/conversation.py @@ -24,7 +24,7 @@ from typing import ClassVar, Literal from gooddata_sdk import GoodDataSdk -from pydantic import BaseModel, Field, ValidationError +from pydantic import BaseModel, Field, ValidationError, model_validator from gooddata_eval.core.agentic._conversation_context import ( CONFIRMATION_REPLY, @@ -100,6 +100,12 @@ class TurnDefinition(BaseModel): over. ``set_answers`` are what the user replies, in order, when the assistant asks a legitimate question on this turn -- the only knowledge the context-mode simulated user has beyond the conversation itself. + + ``user_context`` is the ``userContext`` (dashboard/widget scope) the user has attached + from this turn on, as in the UI: it sticks to later turns until another turn sets it, + and setting it to null clears it. A turn that leaves it out keeps the current one. Its + clarification replies carry it too, since gen-ai treats a message without a context as + cleared. """ turn_id: str @@ -112,6 +118,7 @@ class TurnDefinition(BaseModel): expected_tool_args: dict | None = None depends_on: list[str] = Field(default_factory=list) set_answers: list[str] = Field(default_factory=list) + user_context: dict | None = None class ConversationFixture(BaseModel): @@ -122,6 +129,17 @@ class ConversationFixture(BaseModel): expected_skills: list[str] turns: list[TurnDefinition] + @model_validator(mode="before") + @classmethod + def _reject_fixture_level_user_context(cls, data: object) -> object: + # Unknown keys are ignored, so a context placed here would be dropped without a trace. + if isinstance(data, dict) and "user_context" in data: + raise ValueError( + "user_context is not read from the conversation fixture; put it in the item " + "metadata or input (the context of the first turn), or on a turn" + ) + return data + ReplyRecordKind = Literal["confirmation", "question", "no_action", "turn_incomplete"] @@ -696,6 +714,7 @@ def run_agentic_conversation( mode: ConversationMode | None = None, clarification_judge: ClarificationJudge | None = None, fresh_conversation_per_turn: bool = False, + user_context: dict | None = None, ) -> ConversationResult: """Run a multi-turn, multi-skill conversation evaluation (no K-runs). @@ -706,6 +725,9 @@ def run_agentic_conversation( ``fresh_conversation_per_turn`` sends every turn into a new conversation instead, so the agent sees no history. It is the no-memory baseline a context score is calibrated against: context-dependent turns are expected to fail there. + + ``user_context`` is the context attached when the conversation starts; a turn's own + ``user_context`` replaces it from that turn on. """ resolved_mode = resolve_conversation_mode(mode) if fresh_conversation_per_turn and initial_conversation_id is not None: @@ -747,6 +769,7 @@ def run_agentic_conversation( turn_offset = 0.0 tool_index_offset = 0 reasoning_index_offset = 0 + current_context = user_context try: if initial_conversation_id is not None: @@ -761,6 +784,10 @@ def run_agentic_conversation( conversation_id = client.create_conversation() transcript = [] active_skills = set() + # Before the skip below: a turn whose expectation cannot resolve still changed what + # the user has attached. Only a turn that names the field changes it; null clears it. + if "user_context" in turn.model_fields_set: + current_context = turn.user_context try: resolved_expected = _resolve_refs(turn.expected_output, turn_outputs) resolved_alternatives = [ @@ -813,7 +840,7 @@ def run_agentic_conversation( transcript.append(TranscriptEntry("user", current_message)) incomplete = False try: - chat_result = client.send_message(conversation_id, current_message) + chat_result = client.send_message(conversation_id, current_message, user_context=current_context) except TurnIncompleteError as exc: chat_result = exc.partial_result or ChatResult() incomplete = True @@ -1054,6 +1081,7 @@ def evaluate_agentic_conversation( submit_trace_link: SubmitTraceLink = run_trace_link_inline, mode: ConversationMode | None = None, clarification_judge: ClarificationJudge | None = None, + user_context: dict | None = None, ) -> AgenticEvalOutcome: """Run conversation evaluation, log to Langfuse, and raise on failure. @@ -1078,6 +1106,7 @@ def evaluate_agentic_conversation( agent_id=agent_id, mode=resolved_mode, clarification_judge=clarification_judge, + user_context=user_context, ) passed = result.context_success if resolved_mode == "context" else result.conversation_success diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py index 9e017d57a..1496173b4 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/dashboard_skill.py @@ -955,6 +955,7 @@ def run_agentic_dashboard_skill( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticDashboardSummary: """Run the dashboard-skill agentic evaluation K times and return a summary. @@ -964,7 +965,12 @@ def run_agentic_dashboard_skill( _validate_expectation(expected_output) run_results: list[DashboardRunResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) try: @@ -1021,6 +1027,7 @@ def evaluate_agentic_dashboard_skill( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, gate: EvalGate = DEFAULT_GATE, ) -> AgenticEvalOutcome: """Run dashboard-skill evaluation, log to Langfuse, and raise on failure. @@ -1045,6 +1052,7 @@ def evaluate_agentic_dashboard_skill( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py index 8c8fb3727..dfbbc9e2c 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/general_question.py @@ -116,11 +116,10 @@ def _run_single_general_question( conversation_id: str, question: str, expected_output: str, - user_context: dict | None = None, ) -> GeneralQuestionResult: item_started = time.monotonic() agent_started = time.monotonic() - chat_result = client.send_message(conversation_id, question, user_context=user_context) + chat_result = client.send_message(conversation_id, question) actual_output = render_answer_text(chat_result) agent_elapsed = time.monotonic() - agent_started log_timer( @@ -169,16 +168,19 @@ def run_agentic_general_question( """Run the general-question agentic evaluation K times and return a summary.""" run_results: list[GeneralQuestionResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) judge = LLMJudge(_GENERAL_QUESTION_EVALUATION_STEPS) try: conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() try: - run_results.append( - _run_single_general_question(client, judge, conv_id_0, question, expected_output, user_context) - ) + run_results.append(_run_single_general_question(client, judge, conv_id_0, question, expected_output)) finally: if initial_conversation_id is None: client.delete_conversation(conv_id_0) @@ -186,9 +188,7 @@ def run_agentic_general_question( for _ in range(1, k): conv_id = client.create_conversation() try: - run_results.append( - _run_single_general_question(client, judge, conv_id, question, expected_output, user_context) - ) + run_results.append(_run_single_general_question(client, judge, conv_id, question, expected_output)) finally: client.delete_conversation(conv_id) finally: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py index f3cb43635..790992027 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/guardrail.py @@ -144,11 +144,17 @@ def run_agentic_guardrail( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticGuardrailSummary: """Run the guardrail agentic evaluation K times and return a summary.""" run_results: list[GuardrailResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) judge = LLMJudge(_GUARDRAIL_EVALUATION_STEPS) @@ -206,6 +212,7 @@ def evaluate_agentic_guardrail( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, gate: EvalGate = DEFAULT_GATE, ) -> AgenticEvalOutcome: """Run guardrail evaluation, log to Langfuse, and raise GuardrailAssertionError on failure. @@ -226,6 +233,7 @@ def evaluate_agentic_guardrail( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py index 9e0ced8a6..de87f2c5b 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/kda_skill.py @@ -239,6 +239,7 @@ def run_agentic_kda_skill( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticKdaSummary: """Run the KDA-skill agentic evaluation K times and return a summary. @@ -256,7 +257,12 @@ def run_agentic_kda_skill( raise ValueError(f"k must be >= 1, got {k}") run_results: list[KdaRunResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) def _run_once(conv_id: str) -> KdaRunResult: @@ -424,6 +430,7 @@ def evaluate_agentic_kda_skill( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, gate: EvalGate = DEFAULT_GATE, ) -> AgenticEvalOutcome: """Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure. @@ -444,6 +451,7 @@ def evaluate_agentic_kda_skill( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py index d00abe942..e2e1dced5 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/metric_skill.py @@ -396,6 +396,7 @@ def run_agentic_metric_skill( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticMetricSummary: """Run the metric-skill agentic evaluation K times and return a summary. @@ -405,7 +406,12 @@ def run_agentic_metric_skill( expected_outputs: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output] run_results: list[MetricRunResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) sdk = GoodDataSdk.create(host, token) @@ -467,6 +473,7 @@ def evaluate_agentic_metric_skill( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, gate: EvalGate = DEFAULT_GATE, ) -> AgenticEvalOutcome: """Run metric-skill evaluation, log to Langfuse, and raise MetricSkillAssertionError on failure. @@ -490,6 +497,7 @@ def evaluate_agentic_metric_skill( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py index 75a2e86b0..3e7201311 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py @@ -625,6 +625,7 @@ def run_agentic_report_skill( reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, judge: LLMJudge | None = None, + user_context: dict | None = None, ) -> AgenticReportSummary: """Run the report-skill agentic evaluation K times and return a summary. @@ -639,7 +640,12 @@ def run_agentic_report_skill( judge = LLMJudge(_NARRATIVE_EVALUATION_STEPS) run_results: list[ReportRunResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) try: @@ -698,6 +704,7 @@ def evaluate_agentic_report_skill( submit_trace_link: SubmitTraceLink = run_trace_link_inline, gate: EvalGate = DEFAULT_GATE, judge: LLMJudge | None = None, + user_context: dict | None = None, ) -> AgenticEvalOutcome: """Run report-skill evaluation, log to Langfuse, and raise on failure. @@ -724,6 +731,7 @@ def evaluate_agentic_report_skill( reasoning_effort=reasoning_effort, agent_id=agent_id, judge=judge, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py index ac70deeb7..9194922c0 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/search_tool.py @@ -97,12 +97,18 @@ def run_agentic_search_tool( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticSearchSummary: """Run the search-tool agentic evaluation K times (single-turn each).""" run_results: list[SearchResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) try: conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() @@ -185,6 +191,7 @@ def evaluate_agentic_search_tool( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, gate: EvalGate = DEFAULT_GATE, ) -> AgenticEvalOutcome: """Run search-tool evaluation, log to Langfuse, and raise SearchToolAssertionError on failure. @@ -204,6 +211,7 @@ def evaluate_agentic_search_tool( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py index 51b60f51c..b76b2f1ae 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/visualization.py @@ -301,6 +301,7 @@ def run_agentic_visualization( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticRunSummary: """Run K independent conversations and return evaluation results. @@ -310,7 +311,12 @@ def run_agentic_visualization( conversations created by this function are deleted on completion. """ client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) run_results: list[RunResult] = [] @@ -378,6 +384,7 @@ def evaluate_agentic_visualization( record_output_path: str | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, gate: EvalGate = DEFAULT_GATE, ) -> AgenticEvalOutcome: """Run visualization evaluation, log to Langfuse, and raise VisualizationAssertionError on failure. @@ -400,6 +407,7 @@ def evaluate_agentic_visualization( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py index 9a01b1575..5699764c2 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/what_if.py @@ -293,6 +293,7 @@ def run_agentic_what_if( initial_conversation_id: str | None = None, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict | None = None, ) -> AgenticWhatIfSummary: """Run the what-if agentic evaluation K times and return a summary. @@ -305,7 +306,12 @@ def run_agentic_what_if( raise ValueError(f"k must be >= 1, got {k}") run_results: list[WhatIfRunResult] = [] client = ChatClient( - host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, ) def _run_once(conv_id: str) -> WhatIfRunResult: @@ -472,6 +478,7 @@ def evaluate_agentic_what_if( run_metadata_extra: dict | None = None, reasoning_effort: ReasoningEffort | None = None, submit_trace_link: SubmitTraceLink = run_trace_link_inline, + user_context: dict | None = None, ) -> AgenticEvalOutcome: """Run what-if evaluation, log to Langfuse, and raise WhatIfAssertionError on failure.""" langfuse, window_start = open_trace_window(langfuse) @@ -486,6 +493,7 @@ def evaluate_agentic_what_if( initial_conversation_id=initial_conversation_id, reasoning_effort=reasoning_effort, agent_id=agent_id, + user_context=user_context, ) if langfuse is not None and dataset_item_id: diff --git a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py index a39d9fda5..d8381c907 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py @@ -447,6 +447,7 @@ def __init__( preserve_failed: bool = False, reasoning_effort: ReasoningEffort | None = None, agent_id: str | None = None, + user_context: dict[str, Any] | None = None, ): """Create a chat client bound to one workspace. @@ -455,6 +456,14 @@ def __init__( entirely and the server keeps its own default. The server honours it only while the ``enableGenAiReasoningEffort`` feature flag is on for the organization, so setting it is a request rather than a guarantee. + + ``user_context`` is sent as ``userContext`` on every message unless a call passes + its own. gen-ai scopes each message by the context it carries and treats a message + without one as cleared, so an agentic run's follow-up and clarification turns need + it as much as the first question does. A call cannot clear it, since ``None`` falls + back to it: a caller that changes or clears the context per message builds the + client without one. The server grounds answers in it only while + the ``enableAiContextSetup`` feature flag is on for the organization. """ self._base = f"{host.rstrip('/')}/api/v1/ai/workspaces/{workspace_id}/chat/conversations" self._auth = {"Authorization": f"Bearer {token}"} @@ -474,6 +483,7 @@ def __init__( self._preserve_failed = preserve_failed self._reasoning_effort = normalize_reasoning_effort(reasoning_effort) self._agent_id = agent_id + self._user_context = user_context def create_conversation(self) -> str: def _do() -> str: @@ -507,10 +517,12 @@ def send_message( body: dict[str, Any] = {"item": {"role": "user", "content": {"type": "text", "text": question}}} if self._reasoning_effort is not None: body["options"] = {"reasoningEffort": self._reasoning_effort} - # Only when there is one: gen-ai accepts an explicit null, so assigning - # unconditionally would quietly change every request that has no attachment. - if user_context is not None: - body["userContext"] = user_context + # A per-call context overrides the client's. Only sent when there is one: gen-ai + # accepts an explicit null, so assigning unconditionally would quietly change every + # request that has no attachment. + context = user_context if user_context is not None else self._user_context + if context is not None: + body["userContext"] = context def _do() -> ChatResult: # Set fresh on every retry attempt (before opening this attempt's stream, so its diff --git a/packages/gooddata-eval/src/gooddata_eval/core/dataset/langfuse_source.py b/packages/gooddata-eval/src/gooddata_eval/core/dataset/langfuse_source.py index 6c81ca179..415cc77f2 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/dataset/langfuse_source.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/dataset/langfuse_source.py @@ -66,10 +66,19 @@ def _user_context_from_raw(raw: dict) -> dict[str, Any] | None: item input object or the item metadata. An item carrying an attachment (a WIDGET or VIEW descriptor) is meaningless without it: it degrades into a bare question the agent has no way to answer, and then fails for a reason that has nothing to do with - what the item was written to test. So this has to survive the round trip. + what the item was written to test. So this has to survive the round trip, and a + value that is not an object raises rather than being skipped, for the same reason. """ - found = _first_of(dict, "user_context", raw.get("input"), raw.get("metadata")) - return cast("dict[str, Any]", found) if found is not None else None + for source in (raw.get("input"), raw.get("metadata")): + if not isinstance(source, dict) or source.get("user_context") is None: + continue + found = source["user_context"] + if not isinstance(found, dict): + raise ValueError( + f"Langfuse item {raw.get('id')!r}: user_context must be a JSON object, got {type(found).__name__}" + ) + return cast("dict[str, Any]", found) + return None def _infer_test_kind(expected_output: object, default: str, metadata: object = None) -> str: diff --git a/packages/gooddata-eval/tests/test_agentic_conversation.py b/packages/gooddata-eval/tests/test_agentic_conversation.py index 3d69c0361..4b66e5e9d 100644 --- a/packages/gooddata-eval/tests/test_agentic_conversation.py +++ b/packages/gooddata-eval/tests/test_agentic_conversation.py @@ -869,7 +869,7 @@ def test_run_agentic_conversation_sends_the_next_turn_after_a_self_corrected_ret ) assert mock_client.send_message.call_count == 2 - mock_client.send_message.assert_any_call("conv-1", "Chart it") + mock_client.send_message.assert_any_call("conv-1", "Chart it", user_context=None) assert result.turn_results[0].skill_success is True assert result.turn_results[1].no_error is True assert result.conversation_success is True @@ -1494,12 +1494,13 @@ def _run( client.send_message.side_effect = replies fixture = ConversationFixture(id="c", expected_skills=["visualization"], turns=turns) with ( - patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=client), + patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=client) as client_cls, patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"), ): result = run_agentic_conversation( host="h", token="t", workspace_id="ws", fixture=fixture, mode=mode, clarification_judge=judge, **kw ) + client.constructor_call = client_cls.call_args return result, client @@ -2060,3 +2061,93 @@ def test_context_kept_rate_is_scored_when_there_are_dependent_turns() -> None: scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list} assert scores["context_kept_rate"] == 1.0 assert scores["turns_before_first_break"] == 2 + + +_DASHBOARD_A = {"view": {"dashboard": {"id": "dashboard_000", "title": "Sales Performance Overview"}}} +_DASHBOARD_B = {"view": {"dashboard": {"id": "dashboard_025", "title": "Gross Margin Bridge"}}} + + +def _sent_contexts(client: MagicMock) -> list: + return [c.kwargs["user_context"] for c in client.send_message.call_args_list] + + +def test_user_context_sticks_from_the_turn_that_sets_it_until_another_turn_changes_it() -> None: + turns = [ + _viz_turn("t1", expected=_viz()), + _viz_turn("t2", expected=_viz(), user_context=_DASHBOARD_B), + _viz_turn("t3", expected=_viz()), + _viz_turn("t4", expected=_viz(), user_context=None), + _viz_turn("t5", expected=_viz()), + ] + _, client = _run( + [_result(viz=_viz(), tools=[_SKILLS]) for _ in turns], turns, mode="legacy", user_context=_DASHBOARD_A + ) + # t1 inherits the conversation's context, t3 keeps t2's, and an explicit null clears it. + assert _sent_contexts(client) == [_DASHBOARD_A, _DASHBOARD_B, _DASHBOARD_B, None, None] + + +def test_user_context_rides_on_the_simulated_users_clarification_replies() -> None: + """gen-ai treats a message without a context as cleared, so a reply sent bare would answer + the clarified question with no scope.""" + turns = [_viz_turn("t1", expected=_viz(), user_context=_DASHBOARD_B)] + with patch("gooddata_eval.core.agentic.conversation._get_sim_user_response", return_value="Go ahead."): + _, client = _run( + [_result(text="Which store format do you mean?"), _result(viz=_viz(), tools=[_SKILLS])], + turns, + mode="legacy", + max_clarification_turns=1, + ) + assert _sent_contexts(client) == [_DASHBOARD_B, _DASHBOARD_B] + + +def test_a_skipped_turn_still_changes_the_user_context() -> None: + turns = [ + TurnDefinition( + turn_id="t1", + message="chart it", + expected_skill="visualization", + expected_output={"metrics": ["metric/$ref:missing.metric_id"]}, + user_context=_DASHBOARD_B, + ), + _viz_turn("t2", expected=_viz()), + ] + result, client = _run([_result(viz=_viz(), tools=[_SKILLS])], turns, mode="legacy") + assert result.turn_results[0].mismatches # t1 never ran + assert _sent_contexts(client) == [_DASHBOARD_B] + + +def test_turn_user_context_null_survives_fixture_parsing_as_an_explicit_clear() -> None: + """An explicit null clears the context while an absent key keeps it, so parsing a + Langfuse fixture must tell the two apart.""" + fixture = ConversationFixture.model_validate( + { + "id": "c", + "expected_skills": ["visualization"], + "turns": [ + {"turn_id": "t1", "message": "m", "expected_skill": "visualization", "user_context": None}, + {"turn_id": "t2", "message": "m", "expected_skill": "visualization"}, + ], + } + ) + assert "user_context" in fixture.turns[0].model_fields_set + assert "user_context" not in fixture.turns[1].model_fields_set + + +def test_the_chat_client_is_built_without_a_user_context() -> None: + """A per-call None falls back to the client's context, so one bound at construction + would make a turn's explicit null resend it instead of clearing it.""" + turns = [_viz_turn("t1", expected=_viz(), user_context=None)] + _, client = _run([_result(viz=_viz(), tools=[_SKILLS])], turns, mode="legacy", user_context=_DASHBOARD_A) + assert client.constructor_call.kwargs.get("user_context") is None + + +def test_a_fixture_level_user_context_is_rejected() -> None: + with pytest.raises(ValueError, match="item metadata or input"): + ConversationFixture.model_validate( + { + "id": "c", + "expected_skills": ["visualization"], + "user_context": _DASHBOARD_A, + "turns": [{"turn_id": "t1", "message": "m", "expected_skill": "visualization"}], + } + ) diff --git a/packages/gooddata-eval/tests/test_agentic_general_question.py b/packages/gooddata-eval/tests/test_agentic_general_question.py index 06609f21d..3d265b00a 100644 --- a/packages/gooddata-eval/tests/test_agentic_general_question.py +++ b/packages/gooddata-eval/tests/test_agentic_general_question.py @@ -710,21 +710,3 @@ def test_an_item_with_no_gradeable_run_raises_instead_of_reporting_failures(): # Carried so the runner can still report what the item cost before it became # unevaluable. assert err.value.timings.agent_s == 7.0 # 4.0 + 3.0 - - -def test_run_agentic_general_question_forwards_the_user_context_to_the_chat_client(): - attachment = {"referencedObjects": [{"objects": [{"type": "WIDGET", "id": "campaign_spend"}]}]} - client, judge = _pass_client_and_judge(text_response="It shows campaign spend by channel.") - - with _patched(client, judge): - run_agentic_general_question( - host="https://h", - token="tok", - workspace_id="ws1", - question="What does the visualization I attached show?", - expected_output="Describes the attached chart.", - k=1, - user_context=attachment, - ) - - assert client.send_message.call_args.kwargs["user_context"] == attachment diff --git a/packages/gooddata-eval/tests/test_agentic_runner.py b/packages/gooddata-eval/tests/test_agentic_runner.py index ac893e70b..23586b563 100644 --- a/packages/gooddata-eval/tests/test_agentic_runner.py +++ b/packages/gooddata-eval/tests/test_agentic_runner.py @@ -1,8 +1,10 @@ # (C) 2026 GoodData Corporation. All rights reserved. # SPDX-License-Identifier: LicenseRef-GoodData-Enterprise +import importlib import threading import time from concurrent.futures import ThreadPoolExecutor +from typing import Any from unittest.mock import patch import pytest @@ -752,17 +754,46 @@ def test_an_errored_item_without_timings_keeps_its_zero_defaults(): assert report.items[0].agent_latency_s == 0.0 -def test_dispatch_agentic_passes_user_context_through_to_general_question(): - attachment = {"referencedObjects": [{"objects": [{"type": "WIDGET", "id": "campaign_spend"}]}]} +_ATTACHMENT = {"referencedObjects": [{"objects": [{"type": "WIDGET", "id": "campaign_spend"}]}]} + +# Every agentic kind, with the evaluator it dispatches to and an expected_output it parses. +_KIND_EVALUATORS = [ + ("vis_agentic", "evaluate_agentic_visualization", {"expected_outputs": []}), + ("agentic_visualization", "evaluate_agentic_visualization", {"expected_outputs": []}), + ("agentic_metric_skill", "evaluate_agentic_metric_skill", {}), + ("agentic_dashboard_skill", "evaluate_agentic_dashboard_skill", {}), + ("agentic_alert_skill", "evaluate_agentic_alert_skill", {}), + ("agentic_search", "evaluate_agentic_search_tool", {}), + ("agentic_general_question", "evaluate_agentic_general_question", "Describes the attached chart."), + ("agentic_guardrail", "evaluate_agentic_guardrail", "Refuses."), + ("agentic_kda_skill", "evaluate_agentic_kda_skill", {}), + ("agentic_what_if", "evaluate_agentic_what_if", {}), + ("agentic_anomaly_detection", "evaluate_agentic_anomaly_detection", {}), + ("agentic_report_skill", "evaluate_agentic_report_skill", {}), + ("agentic_conversation", "evaluate_agentic_conversation", {"id": "c1", "expected_skills": [], "turns": []}), +] + + +def test_the_user_context_dispatch_check_covers_every_agentic_kind() -> None: + assert {kind for kind, _, _ in _KIND_EVALUATORS} == AGENTIC_TEST_KINDS + + +@pytest.mark.parametrize(("kind", "evaluator", "expected_output"), _KIND_EVALUATORS) +@pytest.mark.parametrize("user_context", [_ATTACHMENT, None]) +def test_dispatch_agentic_passes_user_context_through_to_every_kind( + kind: str, evaluator: str, expected_output: Any, user_context: dict[str, Any] | None +) -> None: + """A dropped attachment does not fail loudly: the item is asked bare and then fails for + an unrelated reason, so every branch is checked rather than trusted.""" item = DatasetItem( - id="gdai-2179-001", - dataset_name="GDAI-2179", - test_kind="agentic_general_question", + id="i1", + dataset_name="d", + test_kind=kind, question="What does the visualization I attached show?", - expected_output="Describes the attached chart.", - user_context=attachment, + expected_output=expected_output, + user_context=user_context, ) - with patch("gooddata_eval.cli.agentic_runner.evaluate_agentic_general_question") as mock_eval: + with patch(f"gooddata_eval.cli.agentic_runner.{evaluator}") as mock_eval: _dispatch_agentic( item, host="https://h", @@ -773,4 +804,54 @@ def test_dispatch_agentic_passes_user_context_through_to_general_question(): run_ts="2026-01-01", model_version_override=None, ) - assert mock_eval.call_args.kwargs["user_context"] == attachment + assert mock_eval.call_args.kwargs["user_context"] == user_context + + +class _Stop(Exception): + """Raised by a patched collaborator to end a run once the call under test is captured.""" + + +# Kinds whose run_* hands the item's context to its ChatClient, which sends it on every turn. +# conversation passes it per turn instead, covered in its own test file. +_CLIENT_CONTEXT_KINDS = [ + ("visualization", "run_agentic_visualization", "evaluate_agentic_visualization", []), + ("metric_skill", "run_agentic_metric_skill", "evaluate_agentic_metric_skill", {}), + ( + "dashboard_skill", + "run_agentic_dashboard_skill", + "evaluate_agentic_dashboard_skill", + # It validates the expectation before building its client. + {"visualizations": [{"id": "v1", "type": "headline_chart", "title": "v1"}]}, + ), + ("alert_skill", "run_agentic_alert_skill", "evaluate_agentic_alert_skill", {}), + ("search_tool", "run_agentic_search_tool", "evaluate_agentic_search_tool", {}), + ("guardrail", "run_agentic_guardrail", "evaluate_agentic_guardrail", "Refuses."), + ("general_question", "run_agentic_general_question", "evaluate_agentic_general_question", "Describes it."), + ("kda_skill", "run_agentic_kda_skill", "evaluate_agentic_kda_skill", {}), + ("what_if", "run_agentic_what_if", "evaluate_agentic_what_if", {}), + ("anomaly_detection", "run_agentic_anomaly_detection", "evaluate_agentic_anomaly_detection", {}), + ("report_skill", "run_agentic_report_skill", "evaluate_agentic_report_skill", {}), +] + + +@pytest.mark.parametrize(("module_name", "run_fn", "evaluate_fn", "expected"), _CLIENT_CONTEXT_KINDS) +def test_evaluate_agentic_passes_the_user_context_to_its_runner( + module_name: str, run_fn: str, evaluate_fn: str, expected: Any +) -> None: + module = importlib.import_module(f"gooddata_eval.core.agentic.{module_name}") + with patch.object(module, run_fn, side_effect=_Stop) as mock_run, pytest.raises(_Stop): + getattr(module, evaluate_fn)("https://h", "tok", "ws1", "q", expected, user_context=_ATTACHMENT) + assert mock_run.call_args.kwargs["user_context"] == _ATTACHMENT + + +@pytest.mark.parametrize(("module_name", "run_fn", "evaluate_fn", "expected"), _CLIENT_CONTEXT_KINDS) +@pytest.mark.parametrize("user_context", [_ATTACHMENT, None]) +def test_run_agentic_binds_the_user_context_to_its_chat_client( + module_name: str, run_fn: str, evaluate_fn: str, expected: Any, user_context: dict[str, Any] | None +) -> None: + """Bound to the client rather than passed per message, so the follow-up and + clarification turns carry it too -- gen-ai treats a message without one as cleared.""" + module = importlib.import_module(f"gooddata_eval.core.agentic.{module_name}") + with patch.object(module, "ChatClient", side_effect=_Stop) as mock_client, pytest.raises(_Stop): + getattr(module, run_fn)("https://h", "tok", "ws1", "q", expected, user_context=user_context) + assert mock_client.call_args.kwargs["user_context"] == user_context diff --git a/packages/gooddata-eval/tests/test_langfuse_source.py b/packages/gooddata-eval/tests/test_langfuse_source.py index a9db1997e..92448b01e 100644 --- a/packages/gooddata-eval/tests/test_langfuse_source.py +++ b/packages/gooddata-eval/tests/test_langfuse_source.py @@ -163,6 +163,19 @@ def test_item_from_raw_user_context_absent_is_none(): assert item.user_context is None +@pytest.mark.parametrize("value", ["dashboard_000", ["dashboard_000"], 42]) +def test_item_from_raw_rejects_a_user_context_that_is_not_an_object(value: object) -> None: + """Skipped, it would ask the item without its context and fail for an unrelated reason.""" + raw = {**_raw_item("lf-ctx-4", "What does this show?", "PASS if described."), "metadata": {"user_context": value}} + with pytest.raises(ValueError, match="lf-ctx-4.*must be a JSON object"): + _item_from_raw(raw, dataset_name="ds", test_kind="agentic_general_question") + + +def test_item_from_raw_user_context_null_is_none() -> None: + raw = {**_raw_item("lf-ctx-5", "Anything?", "PASS if answered."), "metadata": {"user_context": None}} + assert _item_from_raw(raw, dataset_name="ds", test_kind="agentic_general_question").user_context is None + + def test_item_from_raw_test_kind_from_metadata_beats_default(): """A string expectedOutput gives _infer_test_kind nothing to work with, so a judge-rubric dataset has to carry its kind in metadata or rely on --kind.""" diff --git a/packages/gooddata-eval/tests/test_sse_client.py b/packages/gooddata-eval/tests/test_sse_client.py index c109ac391..e40dae503 100644 --- a/packages/gooddata-eval/tests/test_sse_client.py +++ b/packages/gooddata-eval/tests/test_sse_client.py @@ -1006,6 +1006,29 @@ def test_send_message_omits_user_context_entirely_when_there_is_no_attachment(): assert "userContext" not in captured["body"] +def test_a_client_user_context_goes_on_every_message() -> None: + """An agentic run's follow-up and clarification turns are sent by the same client, and + gen-ai treats a message without a context as cleared.""" + bodies: list[dict] = [] + + def handler(request: httpx.Request) -> httpx.Response: + bodies.append(json.loads(request.read())) + return httpx.Response(200, content=_OK_SSE) + + client = _client_with_handler(handler, user_context=_ATTACHMENT) + client.send_message("conv", "q1") + client.send_message("conv", "q2") + assert [b["userContext"] for b in bodies] == [_ATTACHMENT, _ATTACHMENT] + + +def test_a_per_call_user_context_overrides_the_clients() -> None: + other = {"view": {"dashboard": {"id": "dashboard_000"}}} + captured = {} + client = _client_with_handler(_capture_body(captured), user_context=_ATTACHMENT) + client.send_message("conv", "q", user_context=other) + assert captured["body"]["userContext"] == other + + def test_ask_puts_the_item_attachment_on_the_wire(): """The single-turn path goes through ask(), so an item's user_context has to be forwarded there too, or every non-agentic item with an attachment is asked bare."""