Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
49 changes: 49 additions & 0 deletions packages/gooddata-eval/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -574,6 +574,55 @@ The `expected_output` rubric:
Each criterion is scored independently by the LLM judge, so `quality_score`
is the fraction of satisfied criteria.

### Items asked in a dashboard context (`user_context`)

An item can be asked the way a user asks from an open dashboard or an attached widget.
Its `user_context` is sent verbatim as `userContext` on every chat message of the item,
follow-up and clarification messages included — the server treats a message without one
as a cleared context. Requires the `enableAiContextSetup` feature flag on the target
organization — without it the server ignores the dashboard view. Applies to chat items

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

With the flag off, the server doesn't use the context for grounding, but a dashboard view or attached widgets still trigger the summary skills. Suggest: "Requires the enableAiContextSetup flag. Without it, a view or attached widgets trigger the summary skills instead of grounding."

only; `dashboard_summary` items carry their scope in `summary_input` instead. No
`userContext` key is sent at all when neither the item nor, for `agentic_conversation`,
the current turn sets one.

```json
{
"id": "ctx-001",
"dataset_name": "my_dataset_ctx",
"test_kind": "general_question",
"question": "Which dashboard am I looking at, and what does it cover?",
"user_context": {
"view": {
"dashboard": {
"id": "sales_overview",
"title": "Sales Overview",
"widgets": [
{"widgetType": "insight", "widgetId": "w1", "title": "Revenue by Month", "visualizationId": "revenue_by_month"}
]
}
}
},
"expected_output": "Names the Sales Overview dashboard and summarizes its charts."
}
```

The schema belongs to the AI chat API (`UserContext`: `view.dashboard`,
`referencedObjects`, `activeObject`), so it is not validated here beyond being an object.
In a Langfuse dataset, put it in the item `metadata` (or in the `input` object) under
`user_context`. An object in `input` wins and `metadata` is then not read; a missing or
`null` value in `input` falls back to `metadata`. The value used must be an object or `null`;
anything else fails the dataset load.

`agentic_conversation` items take the item's `user_context` as the context of the first
turn. A turn in `expected_output.turns` can set its own `user_context`, which applies
from that turn on until another turn changes it, as the attached context does in the UI:

| Turn | Context sent |
|---|---|
| no `user_context` key | the current one, unchanged |
| `"user_context": {...}` | this one, from this turn on |
| `"user_context": null` | none, from this turn on |

## Supported test kinds

| test_kind | What the agent must produce | Extra required |
Expand Down
11 changes: 11 additions & 0 deletions packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -182,6 +182,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_metric_skill":
Expand All @@ -194,6 +195,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_dashboard_skill":
Expand All @@ -206,6 +208,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_report_skill":
Expand All @@ -218,6 +221,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_alert_skill":
Expand All @@ -230,6 +234,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_search":
Expand All @@ -245,6 +250,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_general_question":
Expand All @@ -270,6 +276,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_kda_skill":
Expand All @@ -282,6 +289,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_what_if":
Expand All @@ -293,6 +301,7 @@ def _dispatch_agentic(
expected_output=eo if isinstance(eo, dict) else {},
k=k,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_anomaly_detection":
Expand All @@ -305,6 +314,7 @@ def _dispatch_agentic(
k=k,
gate=gate,
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
elif kind == "agentic_conversation":
Expand All @@ -315,6 +325,7 @@ def _dispatch_agentic(
workspace_id=workspace_id,
fixture=ConversationFixture.model_validate(fixture_data),
agent_id=agent_id,
user_context=item.user_context,
**lf_kw,
)
else:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -659,12 +659,18 @@ def run_agentic_alert_skill(
initial_conversation_id: str | None = None,
reasoning_effort: ReasoningEffort | None = None,
agent_id: str | None = None,
user_context: dict | None = None,
) -> AgenticAlertSummary:
"""Run the alert-skill agentic evaluation K times and return a summary."""
expected = _normalize_expected_output(expected_output)
run_results: list[AlertRunResult] = []
client = ChatClient(
host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
host=host,
token=token,
workspace_id=workspace_id,
reasoning_effort=reasoning_effort,
agent_id=agent_id,
user_context=user_context,
)
sdk = GoodDataSdk.create(host, token)

Expand Down Expand Up @@ -858,6 +864,7 @@ def evaluate_agentic_alert_skill(
run_metadata_extra: dict | None = None,
reasoning_effort: ReasoningEffort | None = None,
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
user_context: dict | None = None,
gate: EvalGate = DEFAULT_GATE,
) -> AgenticEvalOutcome:
"""Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.
Expand All @@ -881,6 +888,7 @@ def evaluate_agentic_alert_skill(
initial_conversation_id=initial_conversation_id,
reasoning_effort=reasoning_effort,
agent_id=agent_id,
user_context=user_context,
)

if langfuse is not None and dataset_item_id:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -344,6 +344,7 @@ def run_agentic_anomaly_detection(
initial_conversation_id: str | None = None,
reasoning_effort: ReasoningEffort | None = None,
agent_id: str | None = None,
user_context: dict | None = None,
) -> AgenticAnomalySummary:
"""Run the anomaly-detection agentic evaluation K times and return a summary.

Expand All @@ -356,7 +357,12 @@ def run_agentic_anomaly_detection(
raise ValueError(f"k must be >= 1, got {k}")
run_results: list[AnomalyRunResult] = []
client = ChatClient(
host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
host=host,
token=token,
workspace_id=workspace_id,
reasoning_effort=reasoning_effort,
agent_id=agent_id,
user_context=user_context,
)

def _run_once(conv_id: str) -> AnomalyRunResult:
Expand Down Expand Up @@ -520,6 +526,7 @@ def evaluate_agentic_anomaly_detection(
run_metadata_extra: dict | None = None,
reasoning_effort: ReasoningEffort | None = None,
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
user_context: dict | None = None,
) -> AgenticEvalOutcome:
"""Run anomaly-detection evaluation, log to Langfuse, and raise on failure."""
langfuse, window_start = open_trace_window(langfuse)
Expand All @@ -534,6 +541,7 @@ def evaluate_agentic_anomaly_detection(
initial_conversation_id=initial_conversation_id,
reasoning_effort=reasoning_effort,
agent_id=agent_id,
user_context=user_context,
)

if langfuse is not None and dataset_item_id:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@
from typing import ClassVar, Literal

from gooddata_sdk import GoodDataSdk
from pydantic import BaseModel, Field, ValidationError
from pydantic import BaseModel, Field, ValidationError, model_validator

from gooddata_eval.core.agentic._conversation_context import (
CONFIRMATION_REPLY,
Expand Down Expand Up @@ -100,6 +100,12 @@ class TurnDefinition(BaseModel):
over. ``set_answers`` are what the user replies, in order, when the assistant asks a
legitimate question on this turn -- the only knowledge the context-mode simulated user
has beyond the conversation itself.

``user_context`` is the ``userContext`` (dashboard/widget scope) the user has attached
from this turn on, as in the UI: it sticks to later turns until another turn sets it,
and setting it to null clears it. A turn that leaves it out keeps the current one. Its
clarification replies carry it too, since gen-ai treats a message without a context as
cleared.
"""

turn_id: str
Expand All @@ -112,6 +118,7 @@ class TurnDefinition(BaseModel):
expected_tool_args: dict | None = None
depends_on: list[str] = Field(default_factory=list)
set_answers: list[str] = Field(default_factory=list)
user_context: dict | None = None


class ConversationFixture(BaseModel):
Expand All @@ -122,6 +129,17 @@ class ConversationFixture(BaseModel):
expected_skills: list[str]
turns: list[TurnDefinition]

@model_validator(mode="before")
@classmethod
def _reject_fixture_level_user_context(cls, data: object) -> object:
# Unknown keys are ignored, so a context placed here would be dropped without a trace.
if isinstance(data, dict) and "user_context" in data:
raise ValueError(
"user_context is not read from the conversation fixture; put it in the item "
"metadata or input (the context of the first turn), or on a turn"
)
return data


ReplyRecordKind = Literal["confirmation", "question", "no_action", "turn_incomplete"]

Expand Down Expand Up @@ -696,6 +714,7 @@ def run_agentic_conversation(
mode: ConversationMode | None = None,
clarification_judge: ClarificationJudge | None = None,
fresh_conversation_per_turn: bool = False,
user_context: dict | None = None,
) -> ConversationResult:
"""Run a multi-turn, multi-skill conversation evaluation (no K-runs).

Expand All @@ -706,6 +725,9 @@ def run_agentic_conversation(
``fresh_conversation_per_turn`` sends every turn into a new conversation instead, so the
agent sees no history. It is the no-memory baseline a context score is calibrated
against: context-dependent turns are expected to fail there.

``user_context`` is the context attached when the conversation starts; a turn's own
``user_context`` replaces it from that turn on.
"""
resolved_mode = resolve_conversation_mode(mode)
if fresh_conversation_per_turn and initial_conversation_id is not None:
Expand Down Expand Up @@ -747,6 +769,7 @@ def run_agentic_conversation(
turn_offset = 0.0
tool_index_offset = 0
reasoning_index_offset = 0
current_context = user_context

try:
if initial_conversation_id is not None:
Expand All @@ -761,6 +784,10 @@ def run_agentic_conversation(
conversation_id = client.create_conversation()
transcript = []
active_skills = set()
# Before the skip below: a turn whose expectation cannot resolve still changed what
# the user has attached. Only a turn that names the field changes it; null clears it.
if "user_context" in turn.model_fields_set:
current_context = turn.user_context
try:
resolved_expected = _resolve_refs(turn.expected_output, turn_outputs)
resolved_alternatives = [
Expand Down Expand Up @@ -813,7 +840,7 @@ def run_agentic_conversation(
transcript.append(TranscriptEntry("user", current_message))
incomplete = False
try:
chat_result = client.send_message(conversation_id, current_message)
chat_result = client.send_message(conversation_id, current_message, user_context=current_context)
except TurnIncompleteError as exc:
chat_result = exc.partial_result or ChatResult()
incomplete = True
Expand Down Expand Up @@ -1054,6 +1081,7 @@ def evaluate_agentic_conversation(
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
mode: ConversationMode | None = None,
clarification_judge: ClarificationJudge | None = None,
user_context: dict | None = None,
) -> AgenticEvalOutcome:
"""Run conversation evaluation, log to Langfuse, and raise on failure.

Expand All @@ -1078,6 +1106,7 @@ def evaluate_agentic_conversation(
agent_id=agent_id,
mode=resolved_mode,
clarification_judge=clarification_judge,
user_context=user_context,
)
passed = result.context_success if resolved_mode == "context" else result.conversation_success

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -955,6 +955,7 @@ def run_agentic_dashboard_skill(
initial_conversation_id: str | None = None,
reasoning_effort: ReasoningEffort | None = None,
agent_id: str | None = None,
user_context: dict | None = None,
) -> AgenticDashboardSummary:
"""Run the dashboard-skill agentic evaluation K times and return a summary.

Expand All @@ -964,7 +965,12 @@ def run_agentic_dashboard_skill(
_validate_expectation(expected_output)
run_results: list[DashboardRunResult] = []
client = ChatClient(
host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
host=host,
token=token,
workspace_id=workspace_id,
reasoning_effort=reasoning_effort,
agent_id=agent_id,
user_context=user_context,
)

try:
Expand Down Expand Up @@ -1021,6 +1027,7 @@ def evaluate_agentic_dashboard_skill(
run_metadata_extra: dict | None = None,
reasoning_effort: ReasoningEffort | None = None,
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
user_context: dict | None = None,
gate: EvalGate = DEFAULT_GATE,
) -> AgenticEvalOutcome:
"""Run dashboard-skill evaluation, log to Langfuse, and raise on failure.
Expand All @@ -1045,6 +1052,7 @@ def evaluate_agentic_dashboard_skill(
initial_conversation_id=initial_conversation_id,
reasoning_effort=reasoning_effort,
agent_id=agent_id,
user_context=user_context,
)

if langfuse is not None and dataset_item_id:
Expand Down
Loading
Loading