From 83473b86a08160c50b702fe6dc9bf137dbe64d50 Mon Sep 17 00:00:00 2001 From: Roman Rakus Date: Thu, 8 Oct 2026 14:58:36 +0200 Subject: [PATCH 1/2] refactor(gooddata-eval): name the report skill evaluator for documents Publisher calls its output a document, so the evaluator, its test kind and its check names now say document: agentic_document_skill, document_skill.py, Document* classes and document_* scores. The agentic_report_skill test kind is gone: its one dataset has to be migrated for the Document wire names anyway, and the Tavern shim calls the evaluator function, not the kind. The Python names stay for existing callers: report_skill.py re-exports the old names with a DeprecationWarning, and core.agentic keeps the old aliases. The wire names gen-ai sends are unchanged here; the evaluator still reads the report part. jira: LX-3248 risk: low Co-Authored-By: Claude Opus 5.5 (1M context) --- .../src/gooddata_eval/cli/agentic_runner.py | 12 +- .../gooddata_eval/core/agentic/__init__.py | 30 +- .../core/agentic/document_skill.py | 867 +++++++++++++++++ .../core/agentic/report_skill.py | 897 +----------------- ...kill.py => test_agentic_document_skill.py} | 403 ++++---- .../tests/test_agentic_runner.py | 6 +- .../gooddata-eval/tests/test_trace_linker.py | 2 +- 7 files changed, 1153 insertions(+), 1064 deletions(-) create mode 100644 packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py rename packages/gooddata-eval/tests/{test_agentic_report_skill.py => test_agentic_document_skill.py} (62%) diff --git a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py index 1ead2d118..19edb75c1 100644 --- a/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py +++ b/packages/gooddata-eval/src/gooddata_eval/cli/agentic_runner.py @@ -15,11 +15,11 @@ from gooddata_eval.core.agentic.anomaly_detection import evaluate_agentic_anomaly_detection from gooddata_eval.core.agentic.conversation import ConversationFixture, evaluate_agentic_conversation from gooddata_eval.core.agentic.dashboard_skill import evaluate_agentic_dashboard_skill +from gooddata_eval.core.agentic.document_skill import evaluate_agentic_document_skill from gooddata_eval.core.agentic.general_question import evaluate_agentic_general_question from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill -from gooddata_eval.core.agentic.report_skill import evaluate_agentic_report_skill from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization from gooddata_eval.core.agentic.what_if import evaluate_agentic_what_if @@ -47,7 +47,7 @@ class _LfKw(TypedDict, total=False): "agentic_metric_skill", "agentic_alert_skill", "agentic_dashboard_skill", - "agentic_report_skill", + "agentic_document_skill", "agentic_search", "agentic_general_question", "agentic_guardrail", @@ -98,8 +98,8 @@ class _LfKw(TypedDict, total=False): # # agentic_dashboard_skill is absent by default rather than by evidence: gen-ai holds the draft and # any chart it authors in conversation state and writes neither until a user saves from the UI, so -# it is a candidate for the allowlist once the dataset has runs behind it. agentic_report_skill is -# absent for the same reason: the report draft stays in conversation state until a user saves it. +# it is a candidate for the allowlist once the dataset has runs behind it. agentic_document_skill is +# absent for the same reason: the document draft stays in conversation state until a user saves it. WORKSPACE_MUTATING_TEST_KINDS = frozenset(AGENTIC_TEST_KINDS) - PARALLEL_SAFE_TEST_KINDS @@ -214,8 +214,8 @@ def _dispatch_agentic( user_context=item.user_context, **lf_kw, ) - elif kind == "agentic_report_skill": - return evaluate_agentic_report_skill( + elif kind == "agentic_document_skill": + return evaluate_agentic_document_skill( host=host, token=token, workspace_id=workspace_id, diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py index 9ddbd1b20..b324943de 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/__init__.py @@ -24,6 +24,14 @@ evaluate_agentic_dashboard_skill, run_agentic_dashboard_skill, ) +from gooddata_eval.core.agentic.document_skill import ( + AgenticDocumentSummary, + DocumentEvaluation, + DocumentRunResult, + DocumentSkillAssertionError, + evaluate_agentic_document_skill, + run_agentic_document_skill, +) from gooddata_eval.core.agentic.general_question import ( AgenticGeneralQuestionSummary, GeneralQuestionAssertionError, @@ -53,14 +61,6 @@ evaluate_agentic_metric_skill, run_agentic_metric_skill, ) -from gooddata_eval.core.agentic.report_skill import ( - AgenticReportSummary, - ReportEvaluation, - ReportRunResult, - ReportSkillAssertionError, - evaluate_agentic_report_skill, - run_agentic_report_skill, -) from gooddata_eval.core.agentic.search_tool import ( AgenticSearchSummary, SearchResult, @@ -76,6 +76,14 @@ run_agentic_visualization, ) +# Deprecated names of the document skill, kept for existing callers. +AgenticReportSummary = AgenticDocumentSummary +ReportEvaluation = DocumentEvaluation +ReportRunResult = DocumentRunResult +ReportSkillAssertionError = DocumentSkillAssertionError +evaluate_agentic_report_skill = evaluate_agentic_document_skill +run_agentic_report_skill = run_agentic_document_skill + __all__ = [ "AgenticAlertSummary", "AgenticDashboardSummary", @@ -83,6 +91,7 @@ "AgenticGuardrailSummary", "AgenticKdaSummary", "AgenticMetricSummary", + "AgenticDocumentSummary", "AgenticReportSummary", "AgenticSearchSummary", "AgenticRunSummary", @@ -95,6 +104,9 @@ "DashboardEvaluation", "DashboardRunResult", "DashboardSkillAssertionError", + "DocumentEvaluation", + "DocumentRunResult", + "DocumentSkillAssertionError", "GeneralQuestionAssertionError", "GeneralQuestionResult", "GuardrailAssertionError", @@ -116,6 +128,7 @@ "evaluate_agentic_alert_skill", "evaluate_agentic_conversation", "evaluate_agentic_dashboard_skill", + "evaluate_agentic_document_skill", "evaluate_agentic_general_question", "evaluate_agentic_guardrail", "evaluate_agentic_kda_skill", @@ -126,6 +139,7 @@ "run_agentic_alert_skill", "run_agentic_conversation", "run_agentic_dashboard_skill", + "run_agentic_document_skill", "run_agentic_general_question", "run_agentic_guardrail", "run_agentic_kda_skill", diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py new file mode 100644 index 000000000..fcd58ad12 --- /dev/null +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py @@ -0,0 +1,867 @@ +# (C) 2026 GoodData Corporation. All rights reserved. +"""Agentic document-skill evaluation runner.""" + +from __future__ import annotations + +import re +import time +from collections.abc import Iterator +from dataclasses import dataclass, field +from typing import Any + +from gooddata_eval.core.agentic._conversation_context import classify_reply +from gooddata_eval.core.agentic._failed_runs import build_failed_runs +from gooddata_eval.core.agentic._gate import ( + DEFAULT_GATE, + EvalGate, + gate_failure_note, + gate_passed, + log_gate_scores, + stamp_gate_metadata, +) +from gooddata_eval.core.agentic._trace_linker import ( + RunIdentity, + RunTraceContext, + SubmitTraceLink, + open_trace_window, + run_trace_link_inline, + submit_trace_scoring, + utc_now, +) +from gooddata_eval.core.agentic.dashboard_skill import _extract_tool_result, _skill_activated +from gooddata_eval.core.chat.render import render_answer_text +from gooddata_eval.core.chat.sse_client import ChatClient +from gooddata_eval.core.config import ReasoningEffort +from gooddata_eval.core.evaluators._llm_judge import JudgeResponseError, LLMJudge, score_run +from gooddata_eval.core.models import ( + AgenticAssertionError, + AgenticEvalOutcome, + ChatResult, + ReasoningStepEvent, + ToolCallEvent, + build_latency_breakdown, + shift_and_index_events, +) +from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings + +_DEFAULT_K = 1 +# Same slack as the dashboard skill: the simulated reply is fixed, so the extra rounds only +# absorb a turn that answered without drafting. +_DEFAULT_MAX_ITERATIONS = 4 + +_DRAFT_TOOL = "draft_report" +_BUILDER_SKILL = "report_builder" +_PART_TYPE = "report" +# The copilot's own rule: a document opens with a cover and has at least one content page. +_COVER_PAGE = "cover" +_CONTENT_PAGE = "content" +_EXPECTATION_KEYS = frozenset({"period", "visualizations", "narrative", "expects_clarification"}) +# Template tokens the document renders at export time ({reportName}, {periodStart}, ...). +_PLACEHOLDER_RE = re.compile(r"\{\w+\}") +_SUMMARY_SLOT = "summary" + +_NARRATIVE_EVALUATION_STEPS = [ + "The actual output is a document rendered as text: its title, period, and per page the heading, " + "the chart ids it shows and the written summaries.", + "Check that the summaries address what the INPUT asked the document to cover, as the EXPECTED OUTPUT describes.", + "Check that each page's summary fits that page's heading and charts.", + "Check that the summaries speak to the document's period and do not contradict each other.", + "Fail the output if any summary is placeholder, unfinished or boilerplate text.", +] + + +def _extract_document_part(chat_result: ChatResult) -> dict | None: + """The last ``document`` part of the turn, read back from ``unhandled_parts`` by type. + + The last one, because a turn that drafts and then refines leaves the refined version as + the one the user sees. + """ + for part in reversed(chat_result.unhandled_parts): + if isinstance(part, dict) and part.get("type") == _PART_TYPE: + return part + return None + + +def _nodes(node: Any) -> Iterator[dict]: + """Every node of a page layout in document order, following ``row`` and ``column`` as gen-ai does.""" + if not isinstance(node, dict): + return + yield node + for direction in ("row", "column"): + for child in node.get(direction) or []: + yield from _nodes(child) + + +def _page_kind(page: dict) -> str: + """A page's kind; gen-ai reads a page without one as a content page.""" + return str(page.get("kind") or _CONTENT_PAGE) + + +def _visualizations_of(document: dict) -> set[str]: + """Every visualization id placed anywhere in the document's page layouts.""" + return { + node["visualization"] + for page in document.get("pages") or [] + if isinstance(page, dict) + for node in _nodes(page.get("layout")) + if isinstance(node.get("visualization"), str) + } + + +def _written(text: Any) -> str | None: + """``text`` when it says something once template placeholders are removed, else ``None``.""" + if not isinstance(text, str): + return None + if not re.sub(r"[\W_]+", "", _PLACEHOLDER_RE.sub("", text)): + return None + return text.strip() + + +def _summary_slot(page: dict) -> dict | None: + """The page's ``summary`` slot, as gen-ai reads it, or ``None`` when the layout has none.""" + for node in _nodes(page.get("layout")): + if node.get("id") == _SUMMARY_SLOT and "paragraph" in node: + return node + return None + + +def _summary_text(slot: dict) -> str | None: + """The slot's written text: the model's on an AI summary, the typed string on a static one.""" + paragraph = slot.get("paragraph") + return _written(paragraph.get("text") if isinstance(paragraph, dict) else paragraph) + + +def _content_pages(document: dict) -> list[tuple[int, dict]]: + """``(page number, page)`` for every content page.""" + return [ + (number, page) + for number, page in enumerate(document.get("pages") or [], start=1) + if isinstance(page, dict) and _page_kind(page) == _CONTENT_PAGE + ] + + +def render_document_text(document: dict) -> str: + """The document as the judge reads it: title, period, and per content page its heading, charts and summaries.""" + period = document.get("period") or {} + lines = [f"Document: {document.get('title')}", f"Period: {period.get('start')} to {period.get('end')}"] + for number, page in _content_pages(document): + nodes = list(_nodes(page.get("layout"))) + headings = [h for h in (_written(n.get("heading")) for n in nodes) if h is not None] + charts = [n["visualization"] for n in nodes if isinstance(n.get("visualization"), str)] + lines.append("") + lines.append(f"Page {number}: {', '.join(headings) or '(no heading)'}") + if charts: + lines.append(f"Charts: {', '.join(charts)}") + slot = _summary_slot(page) + text = _summary_text(slot) if slot is not None else None + if text is not None: + lines.append(f"Summary: {text}") + return "\n".join(lines) + + +def _has_narrative(expected_output: dict) -> bool: + return "narrative" in expected_output + + +def _has_period(expected_output: dict) -> bool: + return "period" in expected_output + + +def _has_visualizations(expected_output: dict) -> bool: + return "visualizations" in expected_output + + +def _expects_clarification(expected_output: dict) -> bool: + return "expects_clarification" in expected_output + + +def _validate_expectation(expected_output: Any) -> None: + """Reject a fixture the run could not score meaningfully, before the first API call. + + Every key is optional. A key that is present has to be usable, because a malformed one + would otherwise score vacuously or only surface on the branch where the copilot asks back. + An unknown key is rejected too: a misspelt ``visualisations`` would otherwise leave the + item scored on structure alone. + + Raises: + ValueError: the expectation is unusable. + """ + if not isinstance(expected_output, dict): + raise ValueError(f"expected_output must be an object, got {type(expected_output).__name__}") + unknown = sorted(set(expected_output) - _EXPECTATION_KEYS) + if unknown: + raise ValueError(f"unknown expected_output key(s) {unknown}; known: {sorted(_EXPECTATION_KEYS)}") + if _has_period(expected_output): + period = expected_output.get("period") + if not isinstance(period, dict) or not period.get("start") or not period.get("end"): + raise ValueError(f"period needs both 'start' and 'end', got {period!r}") + if _has_visualizations(expected_output): + visualizations = expected_output.get("visualizations") + if not isinstance(visualizations, list) or not visualizations: + raise ValueError("visualizations is empty; the chart check would pass vacuously") + for entry in visualizations: + if not isinstance(entry, dict) or not entry.get("id"): + raise ValueError(f"every visualization needs an 'id', got {entry!r}") + if _has_narrative(expected_output): + narrative = expected_output.get("narrative") + if not isinstance(narrative, str) or not narrative.strip(): + raise ValueError( + f"narrative must be a non-empty description of what the summaries cover, got {narrative!r}" + ) + if _expects_clarification(expected_output) and not isinstance(expected_output["expects_clarification"], bool): + raise ValueError( + f"expects_clarification must be true or false, got {expected_output['expects_clarification']!r}" + ) + + +def build_simulated_reply(expected_output: dict) -> str: + """The reply the simulated user sends when the copilot asks back instead of drafting. + + Deterministic and LLM-free, built from what the fixture states, so a failure stays the + copilot's rather than a simulated user that phrased things differently each run. + """ + segments: list[str] = [] + visualizations = expected_output.get("visualizations") or [] + if visualizations: + titles = ", ".join(str(v.get("title") or v.get("id")) for v in visualizations) + segments.append(f"Please use these charts: {titles}.") + period = expected_output.get("period") + if isinstance(period, dict): + segments.append(f"Period: {period.get('start')} to {period.get('end')}.") + segments.append("Anything else is up to you. Please create the document now.") + return " ".join(segments) + + +@dataclass(frozen=True) +class _Applies: + """Which conditional checks the case applies; published only when they do. + + A check that could not fail is not evidence, and publishing it as passed would lift + ``quality_score`` above what the run earned. + """ + + period: bool + charts: bool + narrative: bool = False + + +@dataclass +class DocumentEvaluation: + """Per-run outcome of the document-skill checks; ``strict_checks`` is what the run is scored on.""" + + drafted: bool + part_present: bool + ref_matches: bool + pages_consistent: bool + not_saved: bool + skill_activated: bool + applies: _Applies + period_correct: bool = False + charts_matched: bool = False + summaries_present: bool = False + narrative_judged: bool = False + judge_reasoning: str = "" + # Set when the judge returned something unreadable. The run then has no narrative verdict: + # it is neither published as a 0 nor allowed to pass on the remaining checks. + judge_error: str | None = None + failures: list[str] = field(default_factory=list) + + @property + def strict_pass(self) -> bool: + return self.judge_error is None and all(self.strict_checks.values()) + + @property + def ungraded(self) -> bool: + """The judge returned nothing readable and the narrative was the only check still open. + + A run that already failed another check is a failure whatever the judge would have said, + so a judge error there leaves it failed rather than ungraded. + """ + return self.judge_error is not None and all(self.strict_checks.values()) + + @property + def strict_checks(self) -> dict[str, bool]: + # Every name carries the `document_` prefix so a trace can be told apart from other skills' + # by its score names alone; a name shared with another skill would make that ambiguous. + checks = { + "document_drafted": self.drafted, + "document_part_present": self.part_present, + "document_ref_matches": self.ref_matches, + "document_pages_consistent": self.pages_consistent, + "document_not_saved": self.not_saved, + "document_skill_activated": self.skill_activated, + } + if self.applies.period: + checks["document_period_correct"] = self.period_correct + if self.applies.charts: + checks["document_charts_matched"] = self.charts_matched + if self.applies.narrative: + checks["document_summaries_present"] = self.summaries_present + if self.judge_error is None: + checks["document_narrative_judged"] = self.narrative_judged + return checks + + +def _read_document(document_part: dict | None) -> tuple[dict | None, str | None]: + """The part's document, or why it carries no usable one.""" + if document_part is None: + return None, f"the response carries no {_PART_TYPE!r} part" + document = document_part.get("report") + if not isinstance(document, dict): + return ( + None, + f"the {_PART_TYPE!r} part carries no document (report_ref {document_part.get('report_ref')!r})", + ) + if document.get("type") != _PART_TYPE: + return None, ( + f"the {_PART_TYPE!r} part carries a document of type {document.get('type')!r}, expected {_PART_TYPE!r}" + ) + return document, None + + +def evaluate_document_response( + tool_result: dict | None, + document_part: dict | None, + expected_output: dict, + *, + skill_activated: bool, +) -> DocumentEvaluation: + """Score one document response against its expectation. + + Pure: no network and no conversation state, so the whole assertion surface is unit-testable + without an agent. + """ + applies = _Applies( + period=_has_period(expected_output), + charts=_has_visualizations(expected_output), + narrative=_has_narrative(expected_output), + ) + + if tool_result is None: + return DocumentEvaluation( + drafted=False, + part_present=False, + ref_matches=False, + pages_consistent=False, + not_saved=False, + skill_activated=skill_activated, + applies=applies, + failures=[f"the agent never produced a successful {_DRAFT_TOOL} call"], + ) + + document, part_failure = _read_document(document_part) + if document is None or document_part is None: + return DocumentEvaluation( + drafted=True, + part_present=False, + ref_matches=False, + pages_consistent=False, + not_saved=False, + skill_activated=skill_activated, + applies=applies, + failures=[part_failure] if part_failure else [], + ) + + failures: list[str] = [] + + part_ref, tool_ref = document_part.get("report_ref"), tool_result.get("ref") + ref_matches = isinstance(part_ref, str) and bool(part_ref) and part_ref == tool_ref + if not ref_matches: + failures.append(f"the {_PART_TYPE!r} part shows {part_ref!r}, but {_DRAFT_TOOL} returned {tool_ref!r}") + + pages = document.get("pages") or [] + part_count, tool_count = document_part.get("page_count"), tool_result.get("page_count") + if part_count != len(pages) or tool_count != len(pages): + failures.append( + f"the document has {len(pages)} page(s), but the part says {part_count} and {_DRAFT_TOOL} said {tool_count}" + ) + pages_consistent = False + else: + kinds = [_page_kind(page) for page in pages if isinstance(page, dict)] + if not kinds or kinds[0] != _COVER_PAGE: + failures.append(f"the document opens with a {kinds[0] if kinds else None!r} page, not a cover") + if _CONTENT_PAGE not in kinds: + failures.append("the document has no content page") + pages_consistent = bool(kinds) and kinds[0] == _COVER_PAGE and _CONTENT_PAGE in kinds + + not_saved = True + for key, wording in (("saved_report_id", "be saved yet"), ("base_report_id", "edit a saved document")): + value = document_part.get(key) + if value is not None: + failures.append(f"a new draft must not {wording}, but it reports {key} {value!r}") + not_saved = False + + period_correct = False + if applies.period: + expected_period = expected_output["period"] + actual_period = document.get("period") or {} + period_correct = all(actual_period.get(key) == expected_period.get(key) for key in ("start", "end")) + if not period_correct: + failures.append( + f"the document covers {actual_period.get('start')} to {actual_period.get('end')}, " + f"expected {expected_period.get('start')} to {expected_period.get('end')}" + ) + + charts_matched = False + if applies.charts: + placed = _visualizations_of(document) + missing = [v for v in expected_output["visualizations"] if v.get("id") not in placed] + failures.extend(f"the document does not show chart {v.get('id')!r} ({v.get('title')!r})" for v in missing) + charts_matched = not missing + + summaries_present = False + if applies.narrative: + slots = [ + (page, slot) for _number, page in _content_pages(document) if (slot := _summary_slot(page)) is not None + ] + unwritten = [page for page, slot in slots if _summary_text(slot) is None] + if not slots: + failures.append("the document has no summary slot") + failures.extend(f"page {page.get('id')!r} has a summary slot with no written text" for page in unwritten) + summaries_present = bool(slots) and not unwritten + + return DocumentEvaluation( + drafted=True, + part_present=True, + ref_matches=ref_matches, + pages_consistent=pages_consistent, + not_saved=not_saved, + skill_activated=skill_activated, + applies=applies, + period_correct=period_correct, + charts_matched=charts_matched, + summaries_present=summaries_present, + failures=failures, + ) + + +def _judge_narrative( + evaluation: DocumentEvaluation, judge: LLMJudge, document_part: dict | None, question: str, narrative: str +) -> float: + """Grade the drafted document's narrative into ``evaluation``; return the seconds the judge took. + + No document means nothing to grade, which is a failed narrative rather than a judge call. + """ + document = (document_part or {}).get("report") if evaluation.part_present else None + if not isinstance(document, dict): + return 0.0 + started = time.monotonic() + verdict = score_run(judge, input=question, expected_output=narrative, actual_output=render_document_text(document)) + elapsed = time.monotonic() - started + log_timer(f"[timer] document_skill {judge.model} judge complete after {elapsed:.2f}s") + if verdict.error is not None: + evaluation.judge_error = verdict.error + return elapsed + evaluation.narrative_judged = verdict.passed + evaluation.judge_reasoning = verdict.reasoning + if not verdict.passed: + evaluation.failures.append(f"the judge failed the narrative: {verdict.reasoning}") + return elapsed + + +@dataclass +class DocumentRunResult: + """Outcome of one conversation.""" + + conversation_id: str + evaluation: DocumentEvaluation + expects_clarification: bool = False + asked_first: bool = False + tool_result: dict | None = None + document_part: dict | None = None + total_turns: int = 0 + total_steps: int = 0 + reasoning_steps: list[str] = field(default_factory=list) + response_id: str | None = None + tool_call_events: list[ToolCallEvent] = field(default_factory=list) + reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) + timings: PhaseTimings = field(default_factory=PhaseTimings) + + @property + def diagnostics(self) -> dict[str, bool]: + """Observed but not scored. + + Whether the copilot asked before drafting is recorded only when the fixture says it + expects a question, and never gates: how much the copilot should ask is a product + decision, not something a run can get wrong. + """ + return {"document_asked_first": self.asked_first} if self.expects_clarification else {} + + @property + def judge_error(self) -> str | None: + return self.evaluation.judge_error + + @property + def ungraded(self) -> bool: + return self.evaluation.ungraded + + @property + def summaries_from_data(self) -> int | None: + value = (self.tool_result or {}).get("summaries_from_data") + return value if isinstance(value, int) else None + + +@dataclass +class AgenticDocumentSummary: + """Aggregated outcome of K runs.""" + + run_results: list[DocumentRunResult] + pass_at_k: bool + pass_power_k: bool + best: DocumentRunResult + + @property + def scored_run_results(self) -> list[DocumentRunResult]: + """Every run except the ungraded ones: those the narrative verdict alone would have decided.""" + return [r for r in self.run_results if not r.ungraded] + + @property + def judge_errors(self) -> list[str]: + return [r.judge_error for r in self.run_results if r.ungraded and r.judge_error is not None] + + +def _execute_single_document_run( + client: ChatClient, + conversation_id: str, + question: str, + expected_output: dict, + max_iterations: int, + judge: LLMJudge | None = None, +) -> DocumentRunResult: + """Drive one conversation until the copilot drafts a document, then evaluate it. + + Nothing is cleaned up on the way out by design: gen-ai keeps the draft in conversation + state and persists nothing until a user saves it. + """ + tool_result: dict | None = None + document_part: dict | None = None + turns = 0 + steps = 0 + current_question = question + reasoning_steps: list[str] = [] + response_id: str | None = None + all_tool_call_events: list[ToolCallEvent] = [] + all_reasoning_step_events: list[ReasoningStepEvent] = [] + timings = PhaseTimings() + turn_offset = 0.0 + tool_index_offset = 0 + reasoning_index_offset = 0 + first_turn_asked = False + + for iteration in range(max_iterations): + turns += 1 + agent_started = time.monotonic() + chat_result = client.send_message(conversation_id, current_question) + agent_elapsed = time.monotonic() - agent_started + timings.agent_s += agent_elapsed + reasoning_steps.extend(chat_result.reasoning_steps or []) + response_id = chat_result.response_id or response_id + turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events( + chat_result, + turn_offset=turn_offset, + tool_index_offset=tool_index_offset, + reasoning_index_offset=reasoning_index_offset, + ) + all_tool_call_events.extend(chat_result.tool_call_events or []) + all_reasoning_step_events.extend(chat_result.reasoning_step_events or []) + steps += chat_result.reasoning_step_count + + candidate = _extract_tool_result(chat_result.tool_call_events or [], _DRAFT_TOOL) + if candidate is not None: + log_timer( + f"[timer] document_skill {conversation_id} GoodData turn {turns} complete after " + f"{agent_elapsed:.2f}s; {_DRAFT_TOOL} result received" + ) + tool_result = candidate + # Read from the same turn as the draft: the part shows the version that call stored. + document_part = _extract_document_part(chat_result) + break + + response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) + if iteration == 0: + first_turn_asked = classify_reply(chat_result, response_text) == "question" + if not response_text and not chat_result.tool_call_events: + break + if iteration >= max_iterations - 1: + break + log_timer( + f"[timer] document_skill {conversation_id} GoodData turn {turns} complete after " + f"{agent_elapsed:.2f}s; answering with the expected charts and period" + ) + current_question = build_simulated_reply(expected_output) + + evaluation = evaluate_document_response( + tool_result, + document_part, + expected_output, + skill_activated=_skill_activated(all_tool_call_events, _BUILDER_SKILL), + ) + if judge is not None and _has_narrative(expected_output): + timings.judge_s += _judge_narrative(evaluation, judge, document_part, question, expected_output["narrative"]) + + return DocumentRunResult( + conversation_id=conversation_id, + evaluation=evaluation, + expects_clarification=_expects_clarification(expected_output), + asked_first=first_turn_asked, + tool_result=tool_result, + document_part=document_part, + total_turns=turns, + total_steps=steps, + reasoning_steps=reasoning_steps, + response_id=response_id, + tool_call_events=all_tool_call_events, + reasoning_step_events=all_reasoning_step_events, + timings=timings, + ) + + +def run_agentic_document_skill( + host: str, + token: str, + workspace_id: str, + question: str, + expected_output: dict, + k: int = _DEFAULT_K, + max_iterations: int = _DEFAULT_MAX_ITERATIONS, + initial_conversation_id: str | None = None, + reasoning_effort: ReasoningEffort | None = None, + agent_id: str | None = None, + judge: LLMJudge | None = None, + user_context: dict | None = None, +) -> AgenticDocumentSummary: + """Run the document-skill agentic evaluation K times and return a summary. + + The narrative judge is built only for a fixture that states a ``narrative``, so an item + without one needs neither the llm-judge extra nor ``OPENAI_API_KEY``. + + Raises: + ValueError: the fixture is unusable — see ``_validate_expectation``. + """ + _validate_expectation(expected_output) + if judge is None and _has_narrative(expected_output): + judge = LLMJudge(_NARRATIVE_EVALUATION_STEPS) + run_results: list[DocumentRunResult] = [] + client = ChatClient( + host=host, + token=token, + workspace_id=workspace_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + user_context=user_context, + ) + + try: + conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() + try: + run_results.append( + _execute_single_document_run(client, conv_id_0, question, expected_output, max_iterations, judge) + ) + finally: + if initial_conversation_id is None: # only delete conversations we created + client.delete_conversation(conv_id_0) + + for _ in range(1, k): + conv_id = client.create_conversation() + try: + run_results.append( + _execute_single_document_run(client, conv_id, question, expected_output, max_iterations, judge) + ) + finally: + client.delete_conversation(conv_id) + finally: + client.close() + + # An ungraded run is a fault of that judge request, not of the document: it is left out of + # pass@K, and it keeps pass^K from holding, since "every run passed" was never verified. + scored = [r for r in run_results if not r.ungraded] + return AgenticDocumentSummary( + run_results=run_results, + pass_at_k=any(r.evaluation.strict_pass for r in scored), + pass_power_k=len(scored) == len(run_results) and bool(scored) and all(r.evaluation.strict_pass for r in scored), + best=max(scored or run_results, key=lambda r: sum(r.evaluation.strict_checks.values())), + ) + + +class DocumentSkillAssertionError(AgenticAssertionError): + """Raised when a document-skill evaluation fails.""" + + +def evaluate_agentic_document_skill( + host: str, + token: str, + workspace_id: str, + question: str, + expected_output: dict, + k: int = _DEFAULT_K, + max_iterations: int = _DEFAULT_MAX_ITERATIONS, + initial_conversation_id: str | None = None, + agent_id: str | None = None, + langfuse: object | None = None, + dataset_item_id: str = "", + dataset_name: str = "document_skill", + run_timestamp: str | None = None, + model_version_override: str | None = None, + run_metadata_extra: dict | None = None, + reasoning_effort: ReasoningEffort | None = None, + submit_trace_link: SubmitTraceLink = run_trace_link_inline, + gate: EvalGate = DEFAULT_GATE, + judge: LLMJudge | None = None, + user_context: dict | None = None, +) -> AgenticEvalOutcome: + """Run document-skill evaluation, log to Langfuse, and raise on failure. + + Returns the best run's outcome on success; on failure the same values are attached to the + raised ``DocumentSkillAssertionError`` so callers can retrieve them either way. + + Raises: + DocumentSkillAssertionError: the gate did not pass. + ValueError: the fixture is unusable — see ``_validate_expectation``. Raised before any + request, so it means a fixture to fix rather than a result to read. + JudgeResponseError: the fixture states a narrative and the judge returned no readable + verdict for any run -- an item without a result, not K failures. + """ + langfuse, window_start = open_trace_window(langfuse) + summary = run_agentic_document_skill( + host=host, + token=token, + workspace_id=workspace_id, + question=question, + expected_output=expected_output, + k=k, + max_iterations=max_iterations, + initial_conversation_id=initial_conversation_id, + reasoning_effort=reasoning_effort, + agent_id=agent_id, + judge=judge, + user_context=user_context, + ) + + if langfuse is not None and dataset_item_id: + # Pinned on the calling thread: a deferred poll must not widen its query window. + window_end = utc_now() + + def _write_scores(ctx: RunTraceContext) -> None: + stamp_gate_metadata(ctx.run_metadata, k=len(summary.run_results), gate=gate) + + for run_idx, run in enumerate(summary.run_results): + if run.ungraded: + # No verdict decided this run: its other scores would publish a pass the gate + # never counted. + continue + pt = ctx.trace(run.conversation_id) + strict_checks = run.evaluation.strict_checks + with ctx.observe(pt, run_idx, conversation_id=run.conversation_id, output=strict_checks) as tid: + for score_name, value in strict_checks.items(): + ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN") + for name, value in run.diagnostics.items(): + ctx.score(tid, name=name, value=float(value), data_type="BOOLEAN") + if run.summaries_from_data is not None: + ctx.score( + tid, name="document_summaries_from_data", value=run.summaries_from_data, data_type="NUMERIC" + ) + ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC") + ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC") + log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k) + ctx.quality( + tid, + strict_checks=strict_checks, + latency_sec=pt.latency if pt else None, + cost_usd=pt.total_cost if pt else None, + ) + + # Before the pass@K raise: a failing item's scores are the ones worth having. + submit_trace_scoring( + submit_trace_link, + RunIdentity( + host, + token, + workspace_id, + dataset_name, + run_timestamp, + model_version_override, + run_metadata_extra, + reasoning_effort, + ), + langfuse=langfuse, + dataset_item_id=dataset_item_id, + conversation_ids=[r.conversation_id for r in summary.scored_run_results], + window_start=window_start, + window_end=window_end, + suffix_runs=len(summary.run_results) > 1, + write_scores=_write_scores, + item_input=question, + ) + + item_timings = sum_timings([r.timings for r in summary.run_results]) + unscored = summary.judge_errors + if not summary.scored_run_results: + exc_judge = JudgeResponseError( + f"judge returned no readable verdict for any of the {len(summary.run_results)} run(s): " + + " | ".join(unscored) + ) + exc_judge.timings = item_timings + raise exc_judge + + runs_passed = sum(1 for r in summary.run_results if r.evaluation.strict_pass) + runs_effective = len(summary.run_results) + + best = summary.best + + def _run_detail(run: DocumentRunResult) -> dict[str, Any]: + """The diagnostic fields for ONE run, shared by the best run and every failing one. + + A closure because the unscored-run summary belongs to the item, not to the run.""" + return { + **run.evaluation.strict_checks, + **run.diagnostics, + "summaries_from_data": run.summaries_from_data, + "turns": run.total_turns, + **({"judge_reasoning": run.evaluation.judge_reasoning} if run.evaluation.applies.narrative else {}), + "failures": run.evaluation.failures, + "latency_breakdown": build_latency_breakdown(run.tool_call_events, run.reasoning_step_events), + } + + detail: dict[str, Any] = { + **_run_detail(best), + **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), + } + # Same predicate runs_passed is taken over, so an item's failed_runs and its counts + # cannot disagree about which runs failed. + failed_runs = build_failed_runs(summary.run_results, passed=lambda r: r.evaluation.strict_pass, detail=_run_detail) + + if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): + gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) + skill_note = ( + "" + if best.evaluation.skill_activated + else ( + f" set_skills never activated {_BUILDER_SKILL}: either the copilot routed elsewhere, or the skill" + " is not registered. It registers only with enableGenAiReportBuilderSkill and the org's" + " enableBusinessBriefingReportsApp both on." + ) + ) + exc = DocumentSkillAssertionError( + f"Document skill assertion failed. {gate_note}{skill_note} " + f"Checks: {best.evaluation.strict_checks}. " + f"Failures: {'; '.join(best.evaluation.failures) or 'none reported'}." + ) + exc.reasoning_steps = best.reasoning_steps + exc.conversation_id = best.conversation_id + exc.response_id = best.response_id + exc.timings = item_timings + exc.detail = detail + exc.runs_passed = runs_passed + exc.runs_effective = runs_effective + exc.failed_runs = failed_runs + raise exc + return AgenticEvalOutcome( + runs_passed=runs_passed, + runs_effective=runs_effective, + reasoning_steps=best.reasoning_steps, + conversation_id=best.conversation_id, + response_id=best.response_id, + detail=detail, + timings=item_timings, + failed_runs=failed_runs, + ) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py index 86a62bdb0..6220fc931 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/report_skill.py @@ -1,865 +1,50 @@ # (C) 2026 GoodData Corporation. All rights reserved. -"""Agentic report-skill evaluation runner.""" +"""Deprecated: use ``gooddata_eval.core.agentic.document_skill``.""" -from __future__ import annotations +import warnings -import re -import time -from collections.abc import Iterator -from dataclasses import dataclass, field -from typing import Any - -from gooddata_eval.core.agentic._conversation_context import classify_reply -from gooddata_eval.core.agentic._failed_runs import build_failed_runs -from gooddata_eval.core.agentic._gate import ( - DEFAULT_GATE, - EvalGate, - gate_failure_note, - gate_passed, - log_gate_scores, - stamp_gate_metadata, +from gooddata_eval.core.agentic.document_skill import ( + AgenticDocumentSummary as AgenticReportSummary, ) -from gooddata_eval.core.agentic._trace_linker import ( - RunIdentity, - RunTraceContext, - SubmitTraceLink, - open_trace_window, - run_trace_link_inline, - submit_trace_scoring, - utc_now, +from gooddata_eval.core.agentic.document_skill import ( + DocumentEvaluation as ReportEvaluation, ) -from gooddata_eval.core.agentic.dashboard_skill import _extract_tool_result, _skill_activated -from gooddata_eval.core.chat.render import render_answer_text -from gooddata_eval.core.chat.sse_client import ChatClient -from gooddata_eval.core.config import ReasoningEffort -from gooddata_eval.core.evaluators._llm_judge import JudgeResponseError, LLMJudge, score_run -from gooddata_eval.core.models import ( - AgenticAssertionError, - AgenticEvalOutcome, - ChatResult, - ReasoningStepEvent, - ToolCallEvent, - build_latency_breakdown, - shift_and_index_events, +from gooddata_eval.core.agentic.document_skill import ( + DocumentRunResult as ReportRunResult, +) +from gooddata_eval.core.agentic.document_skill import ( + DocumentSkillAssertionError as ReportSkillAssertionError, +) +from gooddata_eval.core.agentic.document_skill import ( + build_simulated_reply, +) +from gooddata_eval.core.agentic.document_skill import ( + evaluate_agentic_document_skill as evaluate_agentic_report_skill, +) +from gooddata_eval.core.agentic.document_skill import ( + evaluate_document_response as evaluate_report_response, +) +from gooddata_eval.core.agentic.document_skill import ( + render_document_text as render_report_text, +) +from gooddata_eval.core.agentic.document_skill import ( + run_agentic_document_skill as run_agentic_report_skill, ) -from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings - -_DEFAULT_K = 1 -# Same slack as the dashboard skill: the simulated reply is fixed, so the extra rounds only -# absorb a turn that answered without drafting. -_DEFAULT_MAX_ITERATIONS = 4 -_DRAFT_TOOL = "draft_report" -_BUILDER_SKILL = "report_builder" -_PART_TYPE = "report" -# The copilot's own rule: a report opens with a cover and has at least one content page. -_COVER_PAGE = "cover" -_CONTENT_PAGE = "content" -_EXPECTATION_KEYS = frozenset({"period", "visualizations", "narrative", "expects_clarification"}) -# Template tokens the report renders at export time ({reportName}, {periodStart}, ...). -_PLACEHOLDER_RE = re.compile(r"\{\w+\}") -_SUMMARY_SLOT = "summary" +warnings.warn( + "gooddata_eval.core.agentic.report_skill is deprecated; import from gooddata_eval.core.agentic.document_skill", + DeprecationWarning, + stacklevel=2, +) -_NARRATIVE_EVALUATION_STEPS = [ - "The actual output is a report rendered as text: its title, period, and per page the heading, " - "the chart ids it shows and the written summaries.", - "Check that the summaries address what the INPUT asked the report to cover, as the EXPECTED OUTPUT describes.", - "Check that each page's summary fits that page's heading and charts.", - "Check that the summaries speak to the report's period and do not contradict each other.", - "Fail the output if any summary is placeholder, unfinished or boilerplate text.", +__all__ = [ + "AgenticReportSummary", + "ReportEvaluation", + "ReportRunResult", + "ReportSkillAssertionError", + "build_simulated_reply", + "evaluate_agentic_report_skill", + "evaluate_report_response", + "render_report_text", + "run_agentic_report_skill", ] - - -def _extract_report_part(chat_result: ChatResult) -> dict | None: - """The last ``report`` part of the turn, read back from ``unhandled_parts`` by type. - - The last one, because a turn that drafts and then refines leaves the refined version as - the one the user sees. - """ - for part in reversed(chat_result.unhandled_parts): - if isinstance(part, dict) and part.get("type") == _PART_TYPE: - return part - return None - - -def _nodes(node: Any) -> Iterator[dict]: - """Every node of a page layout in document order, following ``row`` and ``column`` as gen-ai does.""" - if not isinstance(node, dict): - return - yield node - for direction in ("row", "column"): - for child in node.get(direction) or []: - yield from _nodes(child) - - -def _page_kind(page: dict) -> str: - """A page's kind; gen-ai reads a page without one as a content page.""" - return str(page.get("kind") or _CONTENT_PAGE) - - -def _visualizations_of(report: dict) -> set[str]: - """Every visualization id placed anywhere in the report's page layouts.""" - return { - node["visualization"] - for page in report.get("pages") or [] - if isinstance(page, dict) - for node in _nodes(page.get("layout")) - if isinstance(node.get("visualization"), str) - } - - -def _written(text: Any) -> str | None: - """``text`` when it says something once template placeholders are removed, else ``None``.""" - if not isinstance(text, str): - return None - if not re.sub(r"[\W_]+", "", _PLACEHOLDER_RE.sub("", text)): - return None - return text.strip() - - -def _summary_slot(page: dict) -> dict | None: - """The page's ``summary`` slot, as gen-ai reads it, or ``None`` when the layout has none.""" - for node in _nodes(page.get("layout")): - if node.get("id") == _SUMMARY_SLOT and "paragraph" in node: - return node - return None - - -def _summary_text(slot: dict) -> str | None: - """The slot's written text: the model's on an AI summary, the typed string on a static one.""" - paragraph = slot.get("paragraph") - return _written(paragraph.get("text") if isinstance(paragraph, dict) else paragraph) - - -def _content_pages(report: dict) -> list[tuple[int, dict]]: - """``(page number, page)`` for every content page.""" - return [ - (number, page) - for number, page in enumerate(report.get("pages") or [], start=1) - if isinstance(page, dict) and _page_kind(page) == _CONTENT_PAGE - ] - - -def render_report_text(report: dict) -> str: - """The report as the judge reads it: title, period, and per content page its heading, charts and summaries.""" - period = report.get("period") or {} - lines = [f"Report: {report.get('title')}", f"Period: {period.get('start')} to {period.get('end')}"] - for number, page in _content_pages(report): - nodes = list(_nodes(page.get("layout"))) - headings = [h for h in (_written(n.get("heading")) for n in nodes) if h is not None] - charts = [n["visualization"] for n in nodes if isinstance(n.get("visualization"), str)] - lines.append("") - lines.append(f"Page {number}: {', '.join(headings) or '(no heading)'}") - if charts: - lines.append(f"Charts: {', '.join(charts)}") - slot = _summary_slot(page) - text = _summary_text(slot) if slot is not None else None - if text is not None: - lines.append(f"Summary: {text}") - return "\n".join(lines) - - -def _has_narrative(expected_output: dict) -> bool: - return "narrative" in expected_output - - -def _has_period(expected_output: dict) -> bool: - return "period" in expected_output - - -def _has_visualizations(expected_output: dict) -> bool: - return "visualizations" in expected_output - - -def _expects_clarification(expected_output: dict) -> bool: - return "expects_clarification" in expected_output - - -def _validate_expectation(expected_output: Any) -> None: - """Reject a fixture the run could not score meaningfully, before the first API call. - - Every key is optional. A key that is present has to be usable, because a malformed one - would otherwise score vacuously or only surface on the branch where the copilot asks back. - An unknown key is rejected too: a misspelt ``visualisations`` would otherwise leave the - item scored on structure alone. - - Raises: - ValueError: the expectation is unusable. - """ - if not isinstance(expected_output, dict): - raise ValueError(f"expected_output must be an object, got {type(expected_output).__name__}") - unknown = sorted(set(expected_output) - _EXPECTATION_KEYS) - if unknown: - raise ValueError(f"unknown expected_output key(s) {unknown}; known: {sorted(_EXPECTATION_KEYS)}") - if _has_period(expected_output): - period = expected_output.get("period") - if not isinstance(period, dict) or not period.get("start") or not period.get("end"): - raise ValueError(f"period needs both 'start' and 'end', got {period!r}") - if _has_visualizations(expected_output): - visualizations = expected_output.get("visualizations") - if not isinstance(visualizations, list) or not visualizations: - raise ValueError("visualizations is empty; the chart check would pass vacuously") - for entry in visualizations: - if not isinstance(entry, dict) or not entry.get("id"): - raise ValueError(f"every visualization needs an 'id', got {entry!r}") - if _has_narrative(expected_output): - narrative = expected_output.get("narrative") - if not isinstance(narrative, str) or not narrative.strip(): - raise ValueError( - f"narrative must be a non-empty description of what the summaries cover, got {narrative!r}" - ) - if _expects_clarification(expected_output) and not isinstance(expected_output["expects_clarification"], bool): - raise ValueError( - f"expects_clarification must be true or false, got {expected_output['expects_clarification']!r}" - ) - - -def build_simulated_reply(expected_output: dict) -> str: - """The reply the simulated user sends when the copilot asks back instead of drafting. - - Deterministic and LLM-free, built from what the fixture states, so a failure stays the - copilot's rather than a simulated user that phrased things differently each run. - """ - segments: list[str] = [] - visualizations = expected_output.get("visualizations") or [] - if visualizations: - titles = ", ".join(str(v.get("title") or v.get("id")) for v in visualizations) - segments.append(f"Please use these charts: {titles}.") - period = expected_output.get("period") - if isinstance(period, dict): - segments.append(f"Period: {period.get('start')} to {period.get('end')}.") - segments.append("Anything else is up to you. Please create the report now.") - return " ".join(segments) - - -@dataclass(frozen=True) -class _Applies: - """Which conditional checks the case applies; published only when they do. - - A check that could not fail is not evidence, and publishing it as passed would lift - ``quality_score`` above what the run earned. - """ - - period: bool - charts: bool - narrative: bool = False - - -@dataclass -class ReportEvaluation: - """Per-run outcome of the report-skill checks; ``strict_checks`` is what the run is scored on.""" - - drafted: bool - part_present: bool - ref_matches: bool - pages_consistent: bool - not_saved: bool - skill_activated: bool - applies: _Applies - period_correct: bool = False - charts_matched: bool = False - summaries_present: bool = False - narrative_judged: bool = False - judge_reasoning: str = "" - # Set when the judge returned something unreadable. The run then has no narrative verdict: - # it is neither published as a 0 nor allowed to pass on the remaining checks. - judge_error: str | None = None - failures: list[str] = field(default_factory=list) - - @property - def strict_pass(self) -> bool: - return self.judge_error is None and all(self.strict_checks.values()) - - @property - def ungraded(self) -> bool: - """The judge returned nothing readable and the narrative was the only check still open. - - A run that already failed another check is a failure whatever the judge would have said, - so a judge error there leaves it failed rather than ungraded. - """ - return self.judge_error is not None and all(self.strict_checks.values()) - - @property - def strict_checks(self) -> dict[str, bool]: - # Every name carries the `report_` prefix so a trace can be told apart from other skills' - # by its score names alone; a name shared with another skill would make that ambiguous. - checks = { - "report_drafted": self.drafted, - "report_part_present": self.part_present, - "report_ref_matches": self.ref_matches, - "report_pages_consistent": self.pages_consistent, - "report_not_saved": self.not_saved, - "report_skill_activated": self.skill_activated, - } - if self.applies.period: - checks["report_period_correct"] = self.period_correct - if self.applies.charts: - checks["report_charts_matched"] = self.charts_matched - if self.applies.narrative: - checks["report_summaries_present"] = self.summaries_present - if self.judge_error is None: - checks["report_narrative_judged"] = self.narrative_judged - return checks - - -def _read_report(report_part: dict | None) -> tuple[dict | None, str | None]: - """The part's report document, or why it carries no usable one.""" - if report_part is None: - return None, f"the response carries no {_PART_TYPE!r} part" - report = report_part.get("report") - if not isinstance(report, dict): - return ( - None, - f"the {_PART_TYPE!r} part carries no report document (report_ref {report_part.get('report_ref')!r})", - ) - if report.get("type") != _PART_TYPE: - return None, ( - f"the {_PART_TYPE!r} part carries a document of type {report.get('type')!r}, expected {_PART_TYPE!r}" - ) - return report, None - - -def evaluate_report_response( - tool_result: dict | None, - report_part: dict | None, - expected_output: dict, - *, - skill_activated: bool, -) -> ReportEvaluation: - """Score one report response against its expectation. - - Pure: no network and no conversation state, so the whole assertion surface is unit-testable - without an agent. - """ - applies = _Applies( - period=_has_period(expected_output), - charts=_has_visualizations(expected_output), - narrative=_has_narrative(expected_output), - ) - - if tool_result is None: - return ReportEvaluation( - drafted=False, - part_present=False, - ref_matches=False, - pages_consistent=False, - not_saved=False, - skill_activated=skill_activated, - applies=applies, - failures=[f"the agent never produced a successful {_DRAFT_TOOL} call"], - ) - - report, part_failure = _read_report(report_part) - if report is None or report_part is None: - return ReportEvaluation( - drafted=True, - part_present=False, - ref_matches=False, - pages_consistent=False, - not_saved=False, - skill_activated=skill_activated, - applies=applies, - failures=[part_failure] if part_failure else [], - ) - - failures: list[str] = [] - - part_ref, tool_ref = report_part.get("report_ref"), tool_result.get("ref") - ref_matches = isinstance(part_ref, str) and bool(part_ref) and part_ref == tool_ref - if not ref_matches: - failures.append(f"the {_PART_TYPE!r} part shows {part_ref!r}, but {_DRAFT_TOOL} returned {tool_ref!r}") - - pages = report.get("pages") or [] - part_count, tool_count = report_part.get("page_count"), tool_result.get("page_count") - if part_count != len(pages) or tool_count != len(pages): - failures.append( - f"the report has {len(pages)} page(s), but the part says {part_count} and {_DRAFT_TOOL} said {tool_count}" - ) - pages_consistent = False - else: - kinds = [_page_kind(page) for page in pages if isinstance(page, dict)] - if not kinds or kinds[0] != _COVER_PAGE: - failures.append(f"the report opens with a {kinds[0] if kinds else None!r} page, not a cover") - if _CONTENT_PAGE not in kinds: - failures.append("the report has no content page") - pages_consistent = bool(kinds) and kinds[0] == _COVER_PAGE and _CONTENT_PAGE in kinds - - not_saved = True - for key, wording in (("saved_report_id", "be saved yet"), ("base_report_id", "edit a saved report")): - value = report_part.get(key) - if value is not None: - failures.append(f"a new draft must not {wording}, but it reports {key} {value!r}") - not_saved = False - - period_correct = False - if applies.period: - expected_period = expected_output["period"] - actual_period = report.get("period") or {} - period_correct = all(actual_period.get(key) == expected_period.get(key) for key in ("start", "end")) - if not period_correct: - failures.append( - f"the report covers {actual_period.get('start')} to {actual_period.get('end')}, " - f"expected {expected_period.get('start')} to {expected_period.get('end')}" - ) - - charts_matched = False - if applies.charts: - placed = _visualizations_of(report) - missing = [v for v in expected_output["visualizations"] if v.get("id") not in placed] - failures.extend(f"the report does not show chart {v.get('id')!r} ({v.get('title')!r})" for v in missing) - charts_matched = not missing - - summaries_present = False - if applies.narrative: - slots = [(page, slot) for _number, page in _content_pages(report) if (slot := _summary_slot(page)) is not None] - unwritten = [page for page, slot in slots if _summary_text(slot) is None] - if not slots: - failures.append("the report has no summary slot") - failures.extend(f"page {page.get('id')!r} has a summary slot with no written text" for page in unwritten) - summaries_present = bool(slots) and not unwritten - - return ReportEvaluation( - drafted=True, - part_present=True, - ref_matches=ref_matches, - pages_consistent=pages_consistent, - not_saved=not_saved, - skill_activated=skill_activated, - applies=applies, - period_correct=period_correct, - charts_matched=charts_matched, - summaries_present=summaries_present, - failures=failures, - ) - - -def _judge_narrative( - evaluation: ReportEvaluation, judge: LLMJudge, report_part: dict | None, question: str, narrative: str -) -> float: - """Grade the drafted report's narrative into ``evaluation``; return the seconds the judge took. - - No report document means nothing to grade, which is a failed narrative rather than a judge call. - """ - report = (report_part or {}).get("report") if evaluation.part_present else None - if not isinstance(report, dict): - return 0.0 - started = time.monotonic() - verdict = score_run(judge, input=question, expected_output=narrative, actual_output=render_report_text(report)) - elapsed = time.monotonic() - started - log_timer(f"[timer] report_skill {judge.model} judge complete after {elapsed:.2f}s") - if verdict.error is not None: - evaluation.judge_error = verdict.error - return elapsed - evaluation.narrative_judged = verdict.passed - evaluation.judge_reasoning = verdict.reasoning - if not verdict.passed: - evaluation.failures.append(f"the judge failed the narrative: {verdict.reasoning}") - return elapsed - - -@dataclass -class ReportRunResult: - """Outcome of one conversation.""" - - conversation_id: str - evaluation: ReportEvaluation - expects_clarification: bool = False - asked_first: bool = False - tool_result: dict | None = None - report_part: dict | None = None - total_turns: int = 0 - total_steps: int = 0 - reasoning_steps: list[str] = field(default_factory=list) - response_id: str | None = None - tool_call_events: list[ToolCallEvent] = field(default_factory=list) - reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list) - timings: PhaseTimings = field(default_factory=PhaseTimings) - - @property - def diagnostics(self) -> dict[str, bool]: - """Observed but not scored. - - Whether the copilot asked before drafting is recorded only when the fixture says it - expects a question, and never gates: how much the copilot should ask is a product - decision, not something a run can get wrong. - """ - return {"report_asked_first": self.asked_first} if self.expects_clarification else {} - - @property - def judge_error(self) -> str | None: - return self.evaluation.judge_error - - @property - def ungraded(self) -> bool: - return self.evaluation.ungraded - - @property - def summaries_from_data(self) -> int | None: - value = (self.tool_result or {}).get("summaries_from_data") - return value if isinstance(value, int) else None - - -@dataclass -class AgenticReportSummary: - """Aggregated outcome of K runs.""" - - run_results: list[ReportRunResult] - pass_at_k: bool - pass_power_k: bool - best: ReportRunResult - - @property - def scored_run_results(self) -> list[ReportRunResult]: - """Every run except the ungraded ones: those the narrative verdict alone would have decided.""" - return [r for r in self.run_results if not r.ungraded] - - @property - def judge_errors(self) -> list[str]: - return [r.judge_error for r in self.run_results if r.ungraded and r.judge_error is not None] - - -def _execute_single_report_run( - client: ChatClient, - conversation_id: str, - question: str, - expected_output: dict, - max_iterations: int, - judge: LLMJudge | None = None, -) -> ReportRunResult: - """Drive one conversation until the copilot drafts a report, then evaluate it. - - Nothing is cleaned up on the way out by design: gen-ai keeps the draft in conversation - state and persists nothing until a user saves it. - """ - tool_result: dict | None = None - report_part: dict | None = None - turns = 0 - steps = 0 - current_question = question - reasoning_steps: list[str] = [] - response_id: str | None = None - all_tool_call_events: list[ToolCallEvent] = [] - all_reasoning_step_events: list[ReasoningStepEvent] = [] - timings = PhaseTimings() - turn_offset = 0.0 - tool_index_offset = 0 - reasoning_index_offset = 0 - first_turn_asked = False - - for iteration in range(max_iterations): - turns += 1 - agent_started = time.monotonic() - chat_result = client.send_message(conversation_id, current_question) - agent_elapsed = time.monotonic() - agent_started - timings.agent_s += agent_elapsed - reasoning_steps.extend(chat_result.reasoning_steps or []) - response_id = chat_result.response_id or response_id - turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events( - chat_result, - turn_offset=turn_offset, - tool_index_offset=tool_index_offset, - reasoning_index_offset=reasoning_index_offset, - ) - all_tool_call_events.extend(chat_result.tool_call_events or []) - all_reasoning_step_events.extend(chat_result.reasoning_step_events or []) - steps += chat_result.reasoning_step_count - - candidate = _extract_tool_result(chat_result.tool_call_events or [], _DRAFT_TOOL) - if candidate is not None: - log_timer( - f"[timer] report_skill {conversation_id} GoodData turn {turns} complete after " - f"{agent_elapsed:.2f}s; {_DRAFT_TOOL} result received" - ) - tool_result = candidate - # Read from the same turn as the draft: the part shows the version that call stored. - report_part = _extract_report_part(chat_result) - break - - response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) - if iteration == 0: - first_turn_asked = classify_reply(chat_result, response_text) == "question" - if not response_text and not chat_result.tool_call_events: - break - if iteration >= max_iterations - 1: - break - log_timer( - f"[timer] report_skill {conversation_id} GoodData turn {turns} complete after " - f"{agent_elapsed:.2f}s; answering with the expected charts and period" - ) - current_question = build_simulated_reply(expected_output) - - evaluation = evaluate_report_response( - tool_result, - report_part, - expected_output, - skill_activated=_skill_activated(all_tool_call_events, _BUILDER_SKILL), - ) - if judge is not None and _has_narrative(expected_output): - timings.judge_s += _judge_narrative(evaluation, judge, report_part, question, expected_output["narrative"]) - - return ReportRunResult( - conversation_id=conversation_id, - evaluation=evaluation, - expects_clarification=_expects_clarification(expected_output), - asked_first=first_turn_asked, - tool_result=tool_result, - report_part=report_part, - total_turns=turns, - total_steps=steps, - reasoning_steps=reasoning_steps, - response_id=response_id, - tool_call_events=all_tool_call_events, - reasoning_step_events=all_reasoning_step_events, - timings=timings, - ) - - -def run_agentic_report_skill( - host: str, - token: str, - workspace_id: str, - question: str, - expected_output: dict, - k: int = _DEFAULT_K, - max_iterations: int = _DEFAULT_MAX_ITERATIONS, - initial_conversation_id: str | None = None, - reasoning_effort: ReasoningEffort | None = None, - agent_id: str | None = None, - judge: LLMJudge | None = None, - user_context: dict | None = None, -) -> AgenticReportSummary: - """Run the report-skill agentic evaluation K times and return a summary. - - The narrative judge is built only for a fixture that states a ``narrative``, so an item - without one needs neither the llm-judge extra nor ``OPENAI_API_KEY``. - - Raises: - ValueError: the fixture is unusable — see ``_validate_expectation``. - """ - _validate_expectation(expected_output) - if judge is None and _has_narrative(expected_output): - judge = LLMJudge(_NARRATIVE_EVALUATION_STEPS) - run_results: list[ReportRunResult] = [] - client = ChatClient( - host=host, - token=token, - workspace_id=workspace_id, - reasoning_effort=reasoning_effort, - agent_id=agent_id, - user_context=user_context, - ) - - try: - conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation() - try: - run_results.append( - _execute_single_report_run(client, conv_id_0, question, expected_output, max_iterations, judge) - ) - finally: - if initial_conversation_id is None: # only delete conversations we created - client.delete_conversation(conv_id_0) - - for _ in range(1, k): - conv_id = client.create_conversation() - try: - run_results.append( - _execute_single_report_run(client, conv_id, question, expected_output, max_iterations, judge) - ) - finally: - client.delete_conversation(conv_id) - finally: - client.close() - - # An ungraded run is a fault of that judge request, not of the report: it is left out of - # pass@K, and it keeps pass^K from holding, since "every run passed" was never verified. - scored = [r for r in run_results if not r.ungraded] - return AgenticReportSummary( - run_results=run_results, - pass_at_k=any(r.evaluation.strict_pass for r in scored), - pass_power_k=len(scored) == len(run_results) and bool(scored) and all(r.evaluation.strict_pass for r in scored), - best=max(scored or run_results, key=lambda r: sum(r.evaluation.strict_checks.values())), - ) - - -class ReportSkillAssertionError(AgenticAssertionError): - """Raised when a report-skill evaluation fails.""" - - -def evaluate_agentic_report_skill( - host: str, - token: str, - workspace_id: str, - question: str, - expected_output: dict, - k: int = _DEFAULT_K, - max_iterations: int = _DEFAULT_MAX_ITERATIONS, - initial_conversation_id: str | None = None, - agent_id: str | None = None, - langfuse: object | None = None, - dataset_item_id: str = "", - dataset_name: str = "report_skill", - run_timestamp: str | None = None, - model_version_override: str | None = None, - run_metadata_extra: dict | None = None, - reasoning_effort: ReasoningEffort | None = None, - submit_trace_link: SubmitTraceLink = run_trace_link_inline, - gate: EvalGate = DEFAULT_GATE, - judge: LLMJudge | None = None, - user_context: dict | None = None, -) -> AgenticEvalOutcome: - """Run report-skill evaluation, log to Langfuse, and raise on failure. - - Returns the best run's outcome on success; on failure the same values are attached to the - raised ``ReportSkillAssertionError`` so callers can retrieve them either way. - - Raises: - ReportSkillAssertionError: the gate did not pass. - ValueError: the fixture is unusable — see ``_validate_expectation``. Raised before any - request, so it means a fixture to fix rather than a result to read. - JudgeResponseError: the fixture states a narrative and the judge returned no readable - verdict for any run -- an item without a result, not K failures. - """ - langfuse, window_start = open_trace_window(langfuse) - summary = run_agentic_report_skill( - host=host, - token=token, - workspace_id=workspace_id, - question=question, - expected_output=expected_output, - k=k, - max_iterations=max_iterations, - initial_conversation_id=initial_conversation_id, - reasoning_effort=reasoning_effort, - agent_id=agent_id, - judge=judge, - user_context=user_context, - ) - - if langfuse is not None and dataset_item_id: - # Pinned on the calling thread: a deferred poll must not widen its query window. - window_end = utc_now() - - def _write_scores(ctx: RunTraceContext) -> None: - stamp_gate_metadata(ctx.run_metadata, k=len(summary.run_results), gate=gate) - - for run_idx, run in enumerate(summary.run_results): - if run.ungraded: - # No verdict decided this run: its other scores would publish a pass the gate - # never counted. - continue - pt = ctx.trace(run.conversation_id) - strict_checks = run.evaluation.strict_checks - with ctx.observe(pt, run_idx, conversation_id=run.conversation_id, output=strict_checks) as tid: - for score_name, value in strict_checks.items(): - ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN") - for name, value in run.diagnostics.items(): - ctx.score(tid, name=name, value=float(value), data_type="BOOLEAN") - if run.summaries_from_data is not None: - ctx.score( - tid, name="report_summaries_from_data", value=run.summaries_from_data, data_type="NUMERIC" - ) - ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC") - ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC") - log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k) - ctx.quality( - tid, - strict_checks=strict_checks, - latency_sec=pt.latency if pt else None, - cost_usd=pt.total_cost if pt else None, - ) - - # Before the pass@K raise: a failing item's scores are the ones worth having. - submit_trace_scoring( - submit_trace_link, - RunIdentity( - host, - token, - workspace_id, - dataset_name, - run_timestamp, - model_version_override, - run_metadata_extra, - reasoning_effort, - ), - langfuse=langfuse, - dataset_item_id=dataset_item_id, - conversation_ids=[r.conversation_id for r in summary.scored_run_results], - window_start=window_start, - window_end=window_end, - suffix_runs=len(summary.run_results) > 1, - write_scores=_write_scores, - item_input=question, - ) - - item_timings = sum_timings([r.timings for r in summary.run_results]) - unscored = summary.judge_errors - if not summary.scored_run_results: - exc_judge = JudgeResponseError( - f"judge returned no readable verdict for any of the {len(summary.run_results)} run(s): " - + " | ".join(unscored) - ) - exc_judge.timings = item_timings - raise exc_judge - - runs_passed = sum(1 for r in summary.run_results if r.evaluation.strict_pass) - runs_effective = len(summary.run_results) - - best = summary.best - - def _run_detail(run: ReportRunResult) -> dict[str, Any]: - """The diagnostic fields for ONE run, shared by the best run and every failing one. - - A closure because the unscored-run summary belongs to the item, not to the run.""" - return { - **run.evaluation.strict_checks, - **run.diagnostics, - "summaries_from_data": run.summaries_from_data, - "turns": run.total_turns, - **({"judge_reasoning": run.evaluation.judge_reasoning} if run.evaluation.applies.narrative else {}), - "failures": run.evaluation.failures, - "latency_breakdown": build_latency_breakdown(run.tool_call_events, run.reasoning_step_events), - } - - detail: dict[str, Any] = { - **_run_detail(best), - **({"unscored_runs": len(unscored), "judge_errors": unscored} if unscored else {}), - } - # Same predicate runs_passed is taken over, so an item's failed_runs and its counts - # cannot disagree about which runs failed. - failed_runs = build_failed_runs(summary.run_results, passed=lambda r: r.evaluation.strict_pass, detail=_run_detail) - - if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): - gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - skill_note = ( - "" - if best.evaluation.skill_activated - else ( - f" set_skills never activated {_BUILDER_SKILL}: either the copilot routed elsewhere, or the skill" - " is not registered. It registers only with enableGenAiReportBuilderSkill and the org's" - " enableBusinessBriefingReportsApp both on." - ) - ) - exc = ReportSkillAssertionError( - f"Report skill assertion failed. {gate_note}{skill_note} " - f"Checks: {best.evaluation.strict_checks}. " - f"Failures: {'; '.join(best.evaluation.failures) or 'none reported'}." - ) - exc.reasoning_steps = best.reasoning_steps - exc.conversation_id = best.conversation_id - exc.response_id = best.response_id - exc.timings = item_timings - exc.detail = detail - exc.runs_passed = runs_passed - exc.runs_effective = runs_effective - exc.failed_runs = failed_runs - raise exc - return AgenticEvalOutcome( - runs_passed=runs_passed, - runs_effective=runs_effective, - reasoning_steps=best.reasoning_steps, - conversation_id=best.conversation_id, - response_id=best.response_id, - detail=detail, - timings=item_timings, - failed_runs=failed_runs, - ) diff --git a/packages/gooddata-eval/tests/test_agentic_report_skill.py b/packages/gooddata-eval/tests/test_agentic_document_skill.py similarity index 62% rename from packages/gooddata-eval/tests/test_agentic_report_skill.py rename to packages/gooddata-eval/tests/test_agentic_document_skill.py index d7cb56641..75f4d84b9 100644 --- a/packages/gooddata-eval/tests/test_agentic_report_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_document_skill.py @@ -1,5 +1,6 @@ # (C) 2026 GoodData Corporation. All rights reserved. # SPDX-License-Identifier: LicenseRef-GoodData-Enterprise +import importlib import json from collections.abc import Iterator from contextlib import contextmanager @@ -7,22 +8,23 @@ import pytest from gooddata_eval.cli.agentic_runner import _dispatch_agentic -from gooddata_eval.core.agentic import report_skill -from gooddata_eval.core.agentic.report_skill import ( - ReportEvaluation, - ReportSkillAssertionError, - _execute_single_report_run, +from gooddata_eval.core import agentic +from gooddata_eval.core.agentic import document_skill +from gooddata_eval.core.agentic.document_skill import ( + DocumentEvaluation, + DocumentSkillAssertionError, + _execute_single_document_run, _validate_expectation, _visualizations_of, build_simulated_reply, - evaluate_agentic_report_skill, - evaluate_report_response, - run_agentic_report_skill, + evaluate_agentic_document_skill, + evaluate_document_response, + run_agentic_document_skill, ) from gooddata_eval.core.evaluators._llm_judge import JudgeResponseError from gooddata_eval.core.models import ChatResult, DatasetItem -# Shapes follow what gen-ai writes for a drafted report (composed_report.aac.json): a cover +# Shapes follow what gen-ai writes for a drafted document (composed_report.aac.json): a cover # page, then content pages whose layout nests `column` and `row` entries down to the slots. _REVENUE_TREND = "revenue_trend" _RETURNS_BY_CATEGORY = "returns_by_category" @@ -58,15 +60,15 @@ def _content_page(page_id: str, *visualizations: str) -> dict: } -def _report_part( +def _document_part( pages: list[dict] | None = None, *, - ref: str = "report_1", + ref: str = "document_1", page_count: int | None = None, period: dict | None = None, saved_report_id: str | None = None, base_report_id: str | None = None, - report: dict | None | str = "default", + document: dict | None | str = "default", ) -> dict: pages = pages if pages is not None else [_cover_page(), _content_page("page2", _REVENUE_TREND)] document = ( @@ -77,8 +79,8 @@ def _report_part( "period": period if period is not None else _PERIOD, "pages": pages, } - if report == "default" - else report + if document == "default" + else document ) return { "type": "report", @@ -91,7 +93,7 @@ def _report_part( } -def _draft_result(*, ref: str = "report_1", page_count: int = 2, summaries_from_data: int = 1) -> dict: +def _draft_result(*, ref: str = "document_1", page_count: int = 2, summaries_from_data: int = 1) -> dict: return { "status": "success", "ref": ref, @@ -129,8 +131,8 @@ def _chat_result( def _drafting_turn(**part_kwargs: Any) -> ChatResult: return _chat_result( tool_calls=[_set_skills("report_builder"), _tool_call("draft_report", _draft_result())], - parts=[{"type": "text", "text": "I've put together a report with 2 slides."}, _report_part(**part_kwargs)], - text="I've put together a report with 2 slides.", + parts=[{"type": "text", "text": "I've put together a document with 2 slides."}, _document_part(**part_kwargs)], + text="I've put together a document with 2 slides.", ) @@ -161,7 +163,7 @@ def close(self) -> None: def _install_client(monkeypatch: pytest.MonkeyPatch, client: _ScriptedChatClient) -> None: - monkeypatch.setattr(report_skill, "ChatClient", lambda **_kwargs: client) + monkeypatch.setattr(document_skill, "ChatClient", lambda **_kwargs: client) # ── scoring ───────────────────────────────────────────────────────────────── @@ -169,8 +171,8 @@ def _install_client(monkeypatch: pytest.MonkeyPatch, client: _ScriptedChatClient def _evaluate( part: dict | None, *, expected: dict | None = None, tool: dict | None = None, skill: bool = True -) -> ReportEvaluation: - return evaluate_report_response( +) -> DocumentEvaluation: + return evaluate_document_response( _draft_result() if tool is None else tool, part, {} if expected is None else expected, @@ -178,15 +180,15 @@ def _evaluate( ) -def test_a_drafted_report_returned_in_the_chat_item_passes() -> None: - evaluation = _evaluate(_report_part()) +def test_a_drafted_document_returned_in_the_chat_item_passes() -> None: + evaluation = _evaluate(_document_part()) assert evaluation.strict_checks == { - "report_drafted": True, - "report_part_present": True, - "report_ref_matches": True, - "report_pages_consistent": True, - "report_not_saved": True, - "report_skill_activated": True, + "document_drafted": True, + "document_part_present": True, + "document_ref_matches": True, + "document_pages_consistent": True, + "document_not_saved": True, + "document_skill_activated": True, } assert evaluation.strict_pass assert evaluation.failures == [] @@ -194,92 +196,92 @@ def test_a_drafted_report_returned_in_the_chat_item_passes() -> None: def test_period_and_charts_are_scored_only_when_the_fixture_states_them() -> None: expected = {"period": _PERIOD, "visualizations": [{"id": _REVENUE_TREND, "title": "Revenue trend"}]} - checks = _evaluate(_report_part(), expected=expected).strict_checks - assert checks["report_period_correct"] is True - assert checks["report_charts_matched"] is True + checks = _evaluate(_document_part(), expected=expected).strict_checks + assert checks["document_period_correct"] is True + assert checks["document_charts_matched"] is True def test_no_successful_draft_fails_every_check_and_says_so() -> None: - evaluation = evaluate_report_response(None, None, {"period": _PERIOD}, skill_activated=True) - assert evaluation.strict_checks["report_drafted"] is False - assert evaluation.strict_checks["report_part_present"] is False - assert evaluation.strict_checks["report_period_correct"] is False + evaluation = evaluate_document_response(None, None, {"period": _PERIOD}, skill_activated=True) + assert evaluation.strict_checks["document_drafted"] is False + assert evaluation.strict_checks["document_part_present"] is False + assert evaluation.strict_checks["document_period_correct"] is False assert not evaluation.strict_pass assert evaluation.failures == ["the agent never produced a successful draft_report call"] -def test_a_draft_without_a_report_part_fails() -> None: +def test_a_draft_without_a_document_part_fails() -> None: evaluation = _evaluate(None) - assert evaluation.strict_checks["report_drafted"] is True - assert evaluation.strict_checks["report_part_present"] is False + assert evaluation.strict_checks["document_drafted"] is True + assert evaluation.strict_checks["document_part_present"] is False assert evaluation.failures == ["the response carries no 'report' part"] -def test_a_report_part_whose_document_did_not_resolve_fails() -> None: - evaluation = _evaluate(_report_part(report=None)) - assert evaluation.strict_checks["report_part_present"] is False - assert evaluation.failures == ["the 'report' part carries no report document (report_ref 'report_1')"] +def test_a_document_part_whose_document_did_not_resolve_fails() -> None: + evaluation = _evaluate(_document_part(document=None)) + assert evaluation.strict_checks["document_part_present"] is False + assert evaluation.failures == ["the 'report' part carries no document (report_ref 'document_1')"] -def test_a_document_that_is_not_a_report_fails() -> None: - evaluation = _evaluate(_report_part(report={"id": "x", "type": "dashboard", "pages": []})) - assert evaluation.strict_checks["report_part_present"] is False +def test_a_document_that_is_not_a_document_fails() -> None: + evaluation = _evaluate(_document_part(document={"id": "x", "type": "dashboard", "pages": []})) + assert evaluation.strict_checks["document_part_present"] is False assert evaluation.failures == ["the 'report' part carries a document of type 'dashboard', expected 'report'"] def test_a_part_pointing_at_another_draft_fails() -> None: - evaluation = _evaluate(_report_part(ref="report_2")) - assert evaluation.strict_checks["report_ref_matches"] is False - assert evaluation.failures == ["the 'report' part shows 'report_2', but draft_report returned 'report_1'"] + evaluation = _evaluate(_document_part(ref="document_2")) + assert evaluation.strict_checks["document_ref_matches"] is False + assert evaluation.failures == ["the 'report' part shows 'document_2', but draft_report returned 'document_1'"] def test_a_page_count_that_disagrees_with_the_pages_fails() -> None: - evaluation = _evaluate(_report_part(page_count=3)) - assert evaluation.strict_checks["report_pages_consistent"] is False - assert evaluation.failures == ["the report has 2 page(s), but the part says 3 and draft_report said 2"] + evaluation = _evaluate(_document_part(page_count=3)) + assert evaluation.strict_checks["document_pages_consistent"] is False + assert evaluation.failures == ["the document has 2 page(s), but the part says 3 and draft_report said 2"] -def test_a_cover_alone_is_not_a_report() -> None: - evaluation = _evaluate(_report_part([_cover_page()]), tool=_draft_result(page_count=1)) - assert evaluation.strict_checks["report_pages_consistent"] is False - assert evaluation.failures == ["the report has no content page"] +def test_a_cover_alone_is_not_a_document() -> None: + evaluation = _evaluate(_document_part([_cover_page()]), tool=_draft_result(page_count=1)) + assert evaluation.strict_checks["document_pages_consistent"] is False + assert evaluation.failures == ["the document has no content page"] def test_a_new_draft_must_not_already_be_saved() -> None: - evaluation = _evaluate(_report_part(saved_report_id="sales_overview")) - assert evaluation.strict_checks["report_not_saved"] is False + evaluation = _evaluate(_document_part(saved_report_id="sales_overview")) + assert evaluation.strict_checks["document_not_saved"] is False assert evaluation.failures == ["a new draft must not be saved yet, but it reports saved_report_id 'sales_overview'"] -def test_a_new_draft_must_not_edit_a_saved_report() -> None: - evaluation = _evaluate(_report_part(base_report_id="q3_review")) - assert evaluation.strict_checks["report_not_saved"] is False +def test_a_new_draft_must_not_edit_a_saved_document() -> None: + evaluation = _evaluate(_document_part(base_report_id="q3_review")) + assert evaluation.strict_checks["document_not_saved"] is False assert evaluation.failures == [ - "a new draft must not edit a saved report, but it reports base_report_id 'q3_review'" + "a new draft must not edit a saved document, but it reports base_report_id 'q3_review'" ] def test_a_wrong_period_fails() -> None: evaluation = _evaluate( - _report_part(period={"start": "2025-07-01", "end": "2025-12-31"}), expected={"period": _PERIOD} + _document_part(period={"start": "2025-07-01", "end": "2025-12-31"}), expected={"period": _PERIOD} ) - assert evaluation.strict_checks["report_period_correct"] is False + assert evaluation.strict_checks["document_period_correct"] is False assert evaluation.failures == [ - "the report covers 2025-07-01 to 2025-12-31, expected 2026-01-01 to 2026-06-30", + "the document covers 2025-07-01 to 2025-12-31, expected 2026-01-01 to 2026-06-30", ] def test_a_missing_chart_fails_and_is_named() -> None: expected = {"visualizations": [{"id": _RETURNS_BY_CATEGORY, "title": "Returns by category"}]} - evaluation = _evaluate(_report_part(), expected=expected) - assert evaluation.strict_checks["report_charts_matched"] is False - assert evaluation.failures == ["the report does not show chart 'returns_by_category' ('Returns by category')"] + evaluation = _evaluate(_document_part(), expected=expected) + assert evaluation.strict_checks["document_charts_matched"] is False + assert evaluation.failures == ["the document does not show chart 'returns_by_category' ('Returns by category')"] def test_a_routing_miss_fails_the_skill_check_alone() -> None: - evaluation = _evaluate(_report_part(), skill=False) - assert evaluation.strict_checks["report_skill_activated"] is False - assert [name for name, ok in evaluation.strict_checks.items() if not ok] == ["report_skill_activated"] + evaluation = _evaluate(_document_part(), skill=False) + assert evaluation.strict_checks["document_skill_activated"] is False + assert [name for name, ok in evaluation.strict_checks.items() if not ok] == ["document_skill_activated"] def test_visualizations_are_found_however_deep_the_layout_nests_them() -> None: @@ -323,32 +325,32 @@ def test_the_simulated_reply_names_the_charts_and_the_period() -> None: assert build_simulated_reply(expected) == ( "Please use these charts: Revenue trend, Returns by category. " "Period: 2026-01-01 to 2026-06-30. " - "Anything else is up to you. Please create the report now." + "Anything else is up to you. Please create the document now." ) def test_the_simulated_reply_leaves_out_what_the_fixture_does_not_state() -> None: - assert build_simulated_reply({}) == "Anything else is up to you. Please create the report now." + assert build_simulated_reply({}) == "Anything else is up to you. Please create the document now." # ── conversation loop ─────────────────────────────────────────────────────── -def test_a_report_drafted_on_the_first_turn_ends_the_run() -> None: +def test_a_document_drafted_on_the_first_turn_ends_the_run() -> None: client = _ScriptedChatClient([_drafting_turn()]) - run = _execute_single_report_run(client, "conv-1", "Create a sales report for H1 2026", {}, max_iterations=4) - assert client.sent == ["Create a sales report for H1 2026"] + run = _execute_single_document_run(client, "conv-1", "Create a sales document for H1 2026", {}, max_iterations=4) + assert client.sent == ["Create a sales document for H1 2026"] assert run.total_turns == 1 assert run.asked_first is False assert run.evaluation.strict_pass -def test_a_clarifying_question_is_answered_and_the_report_scored() -> None: +def test_a_clarifying_question_is_answered_and_the_document_scored() -> None: expected = {"period": _PERIOD, "visualizations": [{"id": _REVENUE_TREND, "title": "Revenue trend"}]} - question = _chat_result(text="Which period should the report cover?") + question = _chat_result(text="Which period should the document cover?") client = _ScriptedChatClient([question, _drafting_turn()]) - run = _execute_single_report_run(client, "conv-1", "Make me a report", expected, max_iterations=4) - assert client.sent == ["Make me a report", build_simulated_reply(expected)] + run = _execute_single_document_run(client, "conv-1", "Make me a document", expected, max_iterations=4) + assert client.sent == ["Make me a document", build_simulated_reply(expected)] assert run.total_turns == 2 assert run.asked_first is True assert run.evaluation.strict_pass @@ -356,25 +358,25 @@ def test_a_clarifying_question_is_answered_and_the_report_scored() -> None: def test_a_silent_turn_ends_the_run_without_replying() -> None: client = _ScriptedChatClient([_chat_result()]) - run = _execute_single_report_run(client, "conv-1", "Make me a report", {}, max_iterations=4) - assert client.sent == ["Make me a report"] + run = _execute_single_document_run(client, "conv-1", "Make me a document", {}, max_iterations=4) + assert client.sent == ["Make me a document"] assert not run.evaluation.strict_pass def test_a_copilot_that_never_drafts_stops_at_the_turn_limit() -> None: client = _ScriptedChatClient([_chat_result(text="Which period?")] * 5) - run = _execute_single_report_run(client, "conv-1", "Make me a report", {}, max_iterations=3) + run = _execute_single_document_run(client, "conv-1", "Make me a document", {}, max_iterations=3) assert len(client.sent) == 3 - assert run.evaluation.strict_checks["report_drafted"] is False + assert run.evaluation.strict_checks["document_drafted"] is False def test_asking_first_is_recorded_only_when_the_fixture_expects_it() -> None: - plain = _execute_single_report_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", {}, max_iterations=4) - assert "report_asked_first" not in plain.diagnostics + plain = _execute_single_document_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", {}, max_iterations=4) + assert "document_asked_first" not in plain.diagnostics expected = {"expects_clarification": True} - run = _execute_single_report_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", expected, max_iterations=4) - assert run.diagnostics == {"report_asked_first": False} + run = _execute_single_document_run(_ScriptedChatClient([_drafting_turn()]), "c", "q", expected, max_iterations=4) + assert run.diagnostics == {"document_asked_first": False} assert run.evaluation.strict_pass, "drafting without asking is recorded, never failed" @@ -384,7 +386,7 @@ def test_asking_first_is_recorded_only_when_the_fixture_expects_it() -> None: def test_every_conversation_the_run_creates_is_deleted(monkeypatch: pytest.MonkeyPatch) -> None: client = _ScriptedChatClient([_drafting_turn(), _drafting_turn()]) _install_client(monkeypatch, client) - summary = run_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, k=2) + summary = run_agentic_document_skill("https://h", "tok", "ws", "Create a document", {}, k=2) assert summary.pass_at_k and summary.pass_power_k assert client.deleted == client.created == ["conv-1", "conv-2"] assert client.closed @@ -393,23 +395,23 @@ def test_every_conversation_the_run_creates_is_deleted(monkeypatch: pytest.Monke def test_a_conversation_handed_in_is_not_deleted(monkeypatch: pytest.MonkeyPatch) -> None: client = _ScriptedChatClient([_drafting_turn()]) _install_client(monkeypatch, client) - run_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, initial_conversation_id="theirs") + run_agentic_document_skill("https://h", "tok", "ws", "Create a document", {}, initial_conversation_id="theirs") assert client.deleted == [] def test_a_passing_item_returns_its_checks_and_draft_facts(monkeypatch: pytest.MonkeyPatch) -> None: _install_client(monkeypatch, _ScriptedChatClient([_drafting_turn()])) - outcome = evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {"period": _PERIOD}) + outcome = evaluate_agentic_document_skill("https://h", "tok", "ws", "Create a document", {"period": _PERIOD}) assert outcome.runs_passed == 1 - assert outcome.detail["report_period_correct"] is True + assert outcome.detail["document_period_correct"] is True assert outcome.detail["summaries_from_data"] == 1 assert outcome.detail["failures"] == [] def test_a_failing_item_raises_with_the_failures(monkeypatch: pytest.MonkeyPatch) -> None: _install_client(monkeypatch, _ScriptedChatClient([_chat_result(text="I can't do that.")])) - with pytest.raises(ReportSkillAssertionError) as raised: - evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, max_iterations=1) + with pytest.raises(DocumentSkillAssertionError) as raised: + evaluate_agentic_document_skill("https://h", "tok", "ws", "Create a document", {}, max_iterations=1) assert "the agent never produced a successful draft_report call" in str(raised.value) assert raised.value.runs_passed == 0 assert raised.value.conversation_id == "conv-1" @@ -419,52 +421,53 @@ def test_a_failing_item_without_report_builder_points_at_the_flags(monkeypatch: turn = _chat_result(tool_calls=[_set_skills("visualization")], text="Here is a chart.") _install_client(monkeypatch, _ScriptedChatClient([turn])) with pytest.raises( - ReportSkillAssertionError, match="enableGenAiReportBuilderSkill and the org.s enableBusinessBriefingReportsApp" + DocumentSkillAssertionError, + match="enableGenAiReportBuilderSkill and the org.s enableBusinessBriefingReportsApp", ): - evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {}, max_iterations=1) + evaluate_agentic_document_skill("https://h", "tok", "ws", "Create a document", {}, max_iterations=1) def test_an_unusable_fixture_fails_before_any_request(monkeypatch: pytest.MonkeyPatch) -> None: client = _ScriptedChatClient([]) _install_client(monkeypatch, client) with pytest.raises(ValueError): - evaluate_agentic_report_skill("https://h", "tok", "ws", "Create a report", {"visualizations": []}) + evaluate_agentic_document_skill("https://h", "tok", "ws", "Create a document", {"visualizations": []}) assert client.created == [] def test_a_ref_missing_on_both_sides_does_not_match() -> None: - evaluation = _evaluate(_report_part(ref=None), tool={**_draft_result(), "ref": None}) - assert evaluation.strict_checks["report_ref_matches"] is False + evaluation = _evaluate(_document_part(ref=None), tool={**_draft_result(), "ref": None}) + assert evaluation.strict_checks["document_ref_matches"] is False def test_the_last_successful_draft_of_a_turn_is_the_one_scored() -> None: turn = _chat_result( tool_calls=[ - _tool_call("draft_report", _draft_result(ref="report_1")), + _tool_call("draft_report", _draft_result(ref="document_1")), _tool_call("draft_report", {"status": "error", "message": "no layout fits 7 charts"}), - _tool_call("draft_report", _draft_result(ref="report_2")), + _tool_call("draft_report", _draft_result(ref="document_2")), ], - parts=[_report_part(ref="report_2")], - text="I've put together a report with 2 slides.", + parts=[_document_part(ref="document_2")], + text="I've put together a document with 2 slides.", ) - run = _execute_single_report_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) - assert run.tool_result is not None and run.tool_result["ref"] == "report_2" + run = _execute_single_document_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) + assert run.tool_result is not None and run.tool_result["ref"] == "document_2" assert run.evaluation.strict_pass def test_a_first_turn_that_did_not_ask_is_not_recorded_as_asking() -> None: refusal = _chat_result(text="I could not find any charts on that topic.") expected = {"expects_clarification": True} - run = _execute_single_report_run(_ScriptedChatClient([refusal, _drafting_turn()]), "c", "q", expected, 4) + run = _execute_single_document_run(_ScriptedChatClient([refusal, _drafting_turn()]), "c", "q", expected, 4) assert run.evaluation.strict_pass - assert run.diagnostics == {"report_asked_first": False} + assert run.diagnostics == {"document_asked_first": False} def test_a_structured_clarifying_question_counts_as_asking() -> None: question = _chat_result(parts=[{"type": "clarifyingQuestions", "questions": []}]) expected = {"expects_clarification": True} - run = _execute_single_report_run(_ScriptedChatClient([question, _drafting_turn()]), "c", "q", expected, 4) - assert run.diagnostics == {"report_asked_first": True} + run = _execute_single_document_run(_ScriptedChatClient([question, _drafting_turn()]), "c", "q", expected, 4) + assert run.diagnostics == {"document_asked_first": True} # ── page kinds, fixture keys, asking, two drafts in one turn ─────────────── @@ -479,20 +482,20 @@ def _page(page_id: str, kind: str | None) -> dict: return page -def test_a_report_that_does_not_open_with_a_cover_fails() -> None: - evaluation = _evaluate(_report_part([_page("page1", "content"), _page("page2", "content")])) - assert evaluation.strict_checks["report_pages_consistent"] is False - assert evaluation.failures == ["the report opens with a 'content' page, not a cover"] +def test_a_document_that_does_not_open_with_a_cover_fails() -> None: + evaluation = _evaluate(_document_part([_page("page1", "content"), _page("page2", "content")])) + assert evaluation.strict_checks["document_pages_consistent"] is False + assert evaluation.failures == ["the document opens with a 'content' page, not a cover"] -def test_a_report_without_a_content_page_fails() -> None: - evaluation = _evaluate(_report_part([_cover_page(), _page("page2", "section")])) - assert evaluation.strict_checks["report_pages_consistent"] is False - assert evaluation.failures == ["the report has no content page"] +def test_a_document_without_a_content_page_fails() -> None: + evaluation = _evaluate(_document_part([_cover_page(), _page("page2", "section")])) + assert evaluation.strict_checks["document_pages_consistent"] is False + assert evaluation.failures == ["the document has no content page"] def test_a_page_without_a_kind_is_a_content_page() -> None: - assert _evaluate(_report_part([_cover_page(), _page("page2", None)])).strict_pass + assert _evaluate(_document_part([_cover_page(), _page("page2", None)])).strict_pass @pytest.mark.parametrize("key", ["visualisations", "date_range", "narative"]) @@ -505,7 +508,7 @@ def test_an_expected_output_that_is_not_an_object_is_rejected(monkeypatch: pytes client = _ScriptedChatClient([]) _install_client(monkeypatch, client) item = DatasetItem( - id="i", dataset_name="ds", test_kind="agentic_report_skill", question="q", expected_output="a report" + id="i", dataset_name="ds", test_kind="agentic_document_skill", question="q", expected_output="a document" ) with pytest.raises(ValueError, match="expected_output must be an object"): _dispatch_agentic( @@ -523,28 +526,28 @@ def test_an_expected_output_that_is_not_an_object_is_rejected(monkeypatch: pytes def test_a_copilot_that_only_ever_asks_is_recorded_as_asking() -> None: client = _ScriptedChatClient([_chat_result(text="Which period?")] * 3) - run = _execute_single_report_run(client, "c", "q", {"expects_clarification": True}, max_iterations=2) - assert run.evaluation.strict_checks["report_drafted"] is False - assert run.diagnostics == {"report_asked_first": True} + run = _execute_single_document_run(client, "c", "q", {"expects_clarification": True}, max_iterations=2) + assert run.evaluation.strict_checks["document_drafted"] is False + assert run.diagnostics == {"document_asked_first": True} def test_asking_first_is_recorded_whenever_the_fixture_states_the_key() -> None: - run = _execute_single_report_run( + run = _execute_single_document_run( _ScriptedChatClient([_drafting_turn()]), "c", "q", {"expects_clarification": False}, max_iterations=4 ) - assert run.diagnostics == {"report_asked_first": False} + assert run.diagnostics == {"document_asked_first": False} -def test_the_last_report_part_of_a_turn_is_the_one_scored() -> None: +def test_the_last_document_part_of_a_turn_is_the_one_scored() -> None: turn = _chat_result( tool_calls=[ - _tool_call("draft_report", _draft_result(ref="report_1")), - _tool_call("draft_report", _draft_result(ref="report_2")), + _tool_call("draft_report", _draft_result(ref="document_1")), + _tool_call("draft_report", _draft_result(ref="document_2")), ], - parts=[_report_part(ref="report_1"), _report_part(ref="report_2")], + parts=[_document_part(ref="document_1"), _document_part(ref="document_2")], ) - run = _execute_single_report_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) - assert run.report_part is not None and run.report_part["report_ref"] == "report_2" + run = _execute_single_document_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) + assert run.document_part is not None and run.document_part["report_ref"] == "document_2" assert run.evaluation.strict_pass @@ -582,12 +585,12 @@ def quality( def _scored(monkeypatch: pytest.MonkeyPatch, expected: dict, turns: list[ChatResult], **kwargs: Any) -> _FakeCtx: _install_client(monkeypatch, _ScriptedChatClient(turns)) captured: dict[str, Any] = {} - monkeypatch.setattr(report_skill, "submit_trace_scoring", lambda _link, _identity, **kw: captured.update(kw)) + monkeypatch.setattr(document_skill, "submit_trace_scoring", lambda _link, _identity, **kw: captured.update(kw)) try: - evaluate_agentic_report_skill( + evaluate_agentic_document_skill( "https://h", "tok", "ws", "q", expected, langfuse=object(), dataset_item_id="item-1", **kwargs ) - except ReportSkillAssertionError: + except DocumentSkillAssertionError: pass # scores are written before the gate raises ctx = _FakeCtx() captured["write_scores"](ctx) @@ -598,26 +601,26 @@ def test_every_scored_check_reaches_langfuse_with_the_draft_facts(monkeypatch: p ctx = _scored(monkeypatch, {"period": _PERIOD, "expects_clarification": True}, [_drafting_turn()]) checks = {name for name, kind in ctx.score_types.items() if kind == "BOOLEAN"} assert checks == { - "report_drafted", - "report_part_present", - "report_ref_matches", - "report_pages_consistent", - "report_not_saved", - "report_skill_activated", - "report_period_correct", - "report_asked_first", + "document_drafted", + "document_part_present", + "document_ref_matches", + "document_pages_consistent", + "document_not_saved", + "document_skill_activated", + "document_period_correct", + "document_asked_first", "pass_at_k", "pass_power_k", "gate_passed", } - assert ctx.scores["report_summaries_from_data"] == 1 - assert ctx.score_types["report_summaries_from_data"] == "NUMERIC" - assert "report_asked_first" not in ctx.quality_checks + assert ctx.scores["document_summaries_from_data"] == 1 + assert ctx.score_types["document_summaries_from_data"] == "NUMERIC" + assert "document_asked_first" not in ctx.quality_checks def test_a_check_the_fixture_does_not_state_is_not_published(monkeypatch: pytest.MonkeyPatch) -> None: ctx = _scored(monkeypatch, {}, [_drafting_turn()]) - for absent in ("report_period_correct", "report_charts_matched", "report_asked_first"): + for absent in ("document_period_correct", "document_charts_matched", "document_asked_first"): assert absent not in ctx.scores @@ -655,57 +658,57 @@ def score(self, input: str, expected_output: str, actual_output: str) -> tuple[b def _narrative_turn(*pages: dict) -> ChatResult: return _chat_result( tool_calls=[_tool_call("draft_report", _draft_result(page_count=len(pages)))], - parts=[_report_part(list(pages))], - text="I've put together a report.", + parts=[_document_part(list(pages))], + text="I've put together a document.", ) def test_every_summary_slot_needs_written_text() -> None: pages = [_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND), _summary_page("page3", "")] - evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}, tool=_draft_result(page_count=3)) - assert evaluation.strict_checks["report_summaries_present"] is False + evaluation = _evaluate(_document_part(pages), expected={"narrative": _NARRATIVE}, tool=_draft_result(page_count=3)) + assert evaluation.strict_checks["document_summaries_present"] is False assert evaluation.failures == ["page 'page3' has a summary slot with no written text"] def test_a_content_page_laid_out_without_a_summary_slot_is_fine() -> None: pages = [_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND), _summary_page("page3", None)] - evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}, tool=_draft_result(page_count=3)) - assert evaluation.strict_checks["report_summaries_present"] is True + evaluation = _evaluate(_document_part(pages), expected={"narrative": _NARRATIVE}, tool=_draft_result(page_count=3)) + assert evaluation.strict_checks["document_summaries_present"] is True def test_a_static_text_slot_is_not_a_summary() -> None: page = _summary_page("page2", None, _REVENUE_TREND) page["layout"]["column"].append({"id": "text1", "weight": 1, "paragraph": "Left text."}) - evaluation = _evaluate(_report_part([_cover_page(), page]), expected={"narrative": _NARRATIVE}) - assert evaluation.strict_checks["report_summaries_present"] is False - assert evaluation.failures == ["the report has no summary slot"] + evaluation = _evaluate(_document_part([_cover_page(), page]), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["document_summaries_present"] is False + assert evaluation.failures == ["the document has no summary slot"] def test_a_summary_made_only_of_placeholders_is_not_written() -> None: pages = [_cover_page(), _summary_page("page2", "{periodStart} – {periodEnd}", _REVENUE_TREND)] - evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}) - assert evaluation.strict_checks["report_summaries_present"] is False + evaluation = _evaluate(_document_part(pages), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["document_summaries_present"] is False def test_summaries_are_not_checked_unless_the_fixture_asks_for_the_narrative() -> None: - evaluation = _evaluate(_report_part([_cover_page(), _summary_page("page2", None, _REVENUE_TREND)])) - assert "report_summaries_present" not in evaluation.strict_checks - assert "report_narrative_judged" not in evaluation.strict_checks + evaluation = _evaluate(_document_part([_cover_page(), _summary_page("page2", None, _REVENUE_TREND)])) + assert "document_summaries_present" not in evaluation.strict_checks + assert "document_narrative_judged" not in evaluation.strict_checks -def test_the_judge_reads_the_report_as_text() -> None: +def test_the_judge_reads_the_document_as_text() -> None: judge = _FakeJudge() turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) - run = _execute_single_report_run( - _ScriptedChatClient([turn]), "c", "Report on revenue", {"narrative": _NARRATIVE}, 4, judge=judge + run = _execute_single_document_run( + _ScriptedChatClient([turn]), "c", "Document on revenue", {"narrative": _NARRATIVE}, 4, judge=judge ) - assert run.evaluation.strict_checks["report_narrative_judged"] is True + assert run.evaluation.strict_checks["document_narrative_judged"] is True assert run.evaluation.strict_pass [call] = judge.calls - assert call["input"] == "Report on revenue" + assert call["input"] == "Document on revenue" assert call["expected_output"] == _NARRATIVE assert call["actual_output"] == ( - "Report: Sales overview\n" + "Document: Sales overview\n" "Period: 2026-01-01 to 2026-06-30\n" "\n" "Page 2: Revenue\n" @@ -716,28 +719,28 @@ def test_the_judge_reads_the_report_as_text() -> None: def test_a_failing_verdict_fails_the_run_with_the_judges_reason() -> None: turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) - run = _execute_single_report_run( + run = _execute_single_document_run( _ScriptedChatClient([turn]), "c", "q", {"narrative": _NARRATIVE}, 4, judge=_FakeJudge(passed=False) ) - assert run.evaluation.strict_checks["report_narrative_judged"] is False + assert run.evaluation.strict_checks["document_narrative_judged"] is False assert "the judge failed the narrative: the summaries ignore returns" in run.evaluation.failures -def test_no_report_means_no_judge_call_and_a_failed_narrative() -> None: +def test_no_document_means_no_judge_call_and_a_failed_narrative() -> None: judge = _FakeJudge() - run = _execute_single_report_run( + run = _execute_single_document_run( _ScriptedChatClient([_chat_result(text="Sorry.")]), "c", "q", {"narrative": _NARRATIVE}, 1, judge=judge ) assert judge.calls == [] - assert run.evaluation.strict_checks["report_narrative_judged"] is False + assert run.evaluation.strict_checks["document_narrative_judged"] is False def test_an_unreadable_verdict_leaves_the_run_unscored_not_passed() -> None: judge = _FakeJudge(error=JudgeResponseError("returned no 'score' key")) turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) - run = _execute_single_report_run(_ScriptedChatClient([turn]), "c", "q", {"narrative": _NARRATIVE}, 4, judge=judge) + run = _execute_single_document_run(_ScriptedChatClient([turn]), "c", "q", {"narrative": _NARRATIVE}, 4, judge=judge) assert run.judge_error == "returned no 'score' key" - assert "report_narrative_judged" not in run.evaluation.strict_checks + assert "document_narrative_judged" not in run.evaluation.strict_checks assert not run.evaluation.strict_pass @@ -746,24 +749,24 @@ def test_an_item_the_judge_could_never_grade_raises(monkeypatch: pytest.MonkeyPa _install_client(monkeypatch, _ScriptedChatClient([turn])) judge = _FakeJudge(error=JudgeResponseError("empty body")) with pytest.raises(JudgeResponseError, match="no readable verdict"): - evaluate_agentic_report_skill("https://h", "tok", "ws", "q", {"narrative": _NARRATIVE}, judge=judge) + evaluate_agentic_document_skill("https://h", "tok", "ws", "q", {"narrative": _NARRATIVE}, judge=judge) def test_a_narrative_item_passes_with_its_verdict_in_the_detail(monkeypatch: pytest.MonkeyPatch) -> None: turn = _narrative_turn(_cover_page(), _summary_page("page2", "Revenue grew 12%.", _REVENUE_TREND)) _install_client(monkeypatch, _ScriptedChatClient([turn])) - outcome = evaluate_agentic_report_skill( + outcome = evaluate_agentic_document_skill( "https://h", "tok", "ws", "q", {"narrative": _NARRATIVE}, judge=_FakeJudge() ) - assert outcome.detail["report_narrative_judged"] is True - assert outcome.detail["report_summaries_present"] is True + assert outcome.detail["document_narrative_judged"] is True + assert outcome.detail["document_summaries_present"] is True assert outcome.detail["judge_reasoning"] == "fine" def test_a_fixture_without_a_narrative_never_builds_a_judge(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.delenv("OPENAI_API_KEY", raising=False) _install_client(monkeypatch, _ScriptedChatClient([_drafting_turn()])) - outcome = evaluate_agentic_report_skill("https://h", "tok", "ws", "q", {}) + outcome = evaluate_agentic_document_skill("https://h", "tok", "ws", "q", {}) assert outcome.runs_passed == 1 @@ -772,11 +775,11 @@ def test_an_empty_narrative_is_rejected() -> None: _validate_expectation({"narrative": " "}) -def test_a_report_without_content_pages_has_no_summaries_to_present() -> None: +def test_a_document_without_content_pages_has_no_summaries_to_present() -> None: closing = {**_cover_page(), "id": "page2", "kind": "closing"} - evaluation = _evaluate(_report_part([_cover_page(), closing]), expected={"narrative": _NARRATIVE}) - assert evaluation.strict_checks["report_summaries_present"] is False - assert evaluation.failures == ["the report has no content page", "the report has no summary slot"] + evaluation = _evaluate(_document_part([_cover_page(), closing]), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["document_summaries_present"] is False + assert evaluation.failures == ["the document has no content page", "the document has no summary slot"] class _SequenceJudge(_FakeJudge): @@ -796,7 +799,7 @@ def _narrative_turns() -> list[ChatResult]: def test_an_ungraded_run_keeps_pass_at_k_but_not_pass_power_k(monkeypatch: pytest.MonkeyPatch) -> None: _install_client(monkeypatch, _ScriptedChatClient(_narrative_turns())) - summary = run_agentic_report_skill( + summary = run_agentic_document_skill( "https://h", "tok", "ws", "q", {"narrative": _NARRATIVE}, k=2, judge=_SequenceJudge() ) assert [r.judge_error for r in summary.run_results] == ["empty body", None] @@ -808,8 +811,8 @@ def test_an_ungraded_run_keeps_pass_at_k_but_not_pass_power_k(monkeypatch: pytes def test_an_ungraded_run_writes_no_scores(monkeypatch: pytest.MonkeyPatch) -> None: ctx = _scored(monkeypatch, {"narrative": _NARRATIVE}, _narrative_turns(), k=2, judge=_SequenceJudge()) assert ctx.observed == ["conv-2"] - assert ctx.scores["report_narrative_judged"] == 1.0 - assert ctx.scores["report_summaries_present"] == 1.0 + assert ctx.scores["document_narrative_judged"] == 1.0 + assert ctx.scores["document_summaries_present"] == 1.0 def _failing_narrative_turn() -> ChatResult: @@ -822,22 +825,42 @@ def test_a_run_that_failed_a_fixed_check_is_a_failure_even_when_the_judge_errors expected = {"narrative": _NARRATIVE, "visualizations": [{"id": _RETURNS_BY_CATEGORY, "title": "Returns"}]} _install_client(monkeypatch, _ScriptedChatClient([_failing_narrative_turn()])) judge = _FakeJudge(error=JudgeResponseError("empty body")) - with pytest.raises(ReportSkillAssertionError, match="does not show chart 'returns_by_category'"): - evaluate_agentic_report_skill("https://h", "tok", "ws", "q", expected, judge=judge) + with pytest.raises(DocumentSkillAssertionError, match="does not show chart 'returns_by_category'"): + evaluate_agentic_document_skill("https://h", "tok", "ws", "q", expected, judge=judge) def test_a_failed_run_with_a_judge_error_is_published(monkeypatch: pytest.MonkeyPatch) -> None: expected = {"narrative": _NARRATIVE, "visualizations": [{"id": _RETURNS_BY_CATEGORY, "title": "Returns"}]} ctx = _scored(monkeypatch, expected, [_failing_narrative_turn()], judge=_FakeJudge(error=JudgeResponseError("x"))) assert ctx.observed == ["conv-1"] - assert ctx.scores["report_charts_matched"] == 0.0 - assert "report_narrative_judged" not in ctx.scores + assert ctx.scores["document_charts_matched"] == 0.0 + assert "document_narrative_judged" not in ctx.scores def test_only_content_pages_count_for_summaries() -> None: cover = _cover_page() cover["layout"]["column"].append({"id": "summary", "weight": 1, "paragraph": {"text": "Revenue grew 12%."}}) pages = [cover, _summary_page("page2", None, _REVENUE_TREND)] - evaluation = _evaluate(_report_part(pages), expected={"narrative": _NARRATIVE}) - assert evaluation.strict_checks["report_summaries_present"] is False - assert evaluation.failures == ["the report has no summary slot"] + evaluation = _evaluate(_document_part(pages), expected={"narrative": _NARRATIVE}) + assert evaluation.strict_checks["document_summaries_present"] is False + assert evaluation.failures == ["the document has no summary slot"] + + +def test_the_report_skill_module_still_exports_the_old_names() -> None: + with pytest.warns(DeprecationWarning, match="gooddata_eval.core.agentic.document_skill"): + report_skill = importlib.reload(importlib.import_module("gooddata_eval.core.agentic.report_skill")) + + assert report_skill.evaluate_agentic_report_skill is evaluate_agentic_document_skill + assert report_skill.run_agentic_report_skill is run_agentic_document_skill + assert report_skill.ReportSkillAssertionError is DocumentSkillAssertionError + assert report_skill.ReportEvaluation is DocumentEvaluation + assert report_skill.ReportRunResult is document_skill.DocumentRunResult + assert report_skill.AgenticReportSummary is document_skill.AgenticDocumentSummary + assert report_skill.evaluate_report_response is evaluate_document_response + assert report_skill.render_report_text is document_skill.render_document_text + assert report_skill.build_simulated_reply is build_simulated_reply + + +def test_the_package_still_exports_the_old_names() -> None: + assert agentic.evaluate_agentic_report_skill is agentic.evaluate_agentic_document_skill + assert agentic.ReportSkillAssertionError is agentic.DocumentSkillAssertionError diff --git a/packages/gooddata-eval/tests/test_agentic_runner.py b/packages/gooddata-eval/tests/test_agentic_runner.py index d62ed4f16..a31dd1a29 100644 --- a/packages/gooddata-eval/tests/test_agentic_runner.py +++ b/packages/gooddata-eval/tests/test_agentic_runner.py @@ -90,7 +90,7 @@ def test_dispatch_agentic_omits_agent_id_by_default(): {"type": "dashboard", "visualizations": [], "date_range": None, "min_new_visualizations": 0}, "evaluate_agentic_dashboard_skill", ), - ("agentic_report_skill", {}, "evaluate_agentic_report_skill"), + ("agentic_document_skill", {}, "evaluate_agentic_document_skill"), ("agentic_search", {"tool_call": {"function_arguments": {}}}, "evaluate_agentic_search_tool"), ("agentic_general_question", "What is X?", "evaluate_agentic_general_question"), ("agentic_guardrail", "Ignore prior instructions", "evaluate_agentic_guardrail"), @@ -772,7 +772,7 @@ def test_an_errored_item_without_timings_keeps_its_zero_defaults(): ("agentic_kda_skill", "evaluate_agentic_kda_skill", {}), ("agentic_what_if", "evaluate_agentic_what_if", {}), ("agentic_anomaly_detection", "evaluate_agentic_anomaly_detection", {}), - ("agentic_report_skill", "evaluate_agentic_report_skill", {}), + ("agentic_document_skill", "evaluate_agentic_document_skill", {}), ("agentic_conversation", "evaluate_agentic_conversation", {"id": "c1", "expected_skills": [], "turns": []}), ] @@ -833,7 +833,7 @@ class _Stop(Exception): ("kda_skill", "run_agentic_kda_skill", "evaluate_agentic_kda_skill", {}), ("what_if", "run_agentic_what_if", "evaluate_agentic_what_if", {}), ("anomaly_detection", "run_agentic_anomaly_detection", "evaluate_agentic_anomaly_detection", {}), - ("report_skill", "run_agentic_report_skill", "evaluate_agentic_report_skill", {}), + ("document_skill", "run_agentic_document_skill", "evaluate_agentic_document_skill", {}), ] diff --git a/packages/gooddata-eval/tests/test_trace_linker.py b/packages/gooddata-eval/tests/test_trace_linker.py index 9baabb6fe..7c7c0e0e1 100644 --- a/packages/gooddata-eval/tests/test_trace_linker.py +++ b/packages/gooddata-eval/tests/test_trace_linker.py @@ -115,7 +115,7 @@ def test_run_trace_link_inline_runs_the_task_on_the_calling_thread(): ("metric_skill", "evaluate_agentic_metric_skill"), ("alert_skill", "evaluate_agentic_alert_skill"), ("dashboard_skill", "evaluate_agentic_dashboard_skill"), - ("report_skill", "evaluate_agentic_report_skill"), + ("document_skill", "evaluate_agentic_document_skill"), ("search_tool", "evaluate_agentic_search_tool"), ("visualization", "evaluate_agentic_visualization"), ("kda_skill", "evaluate_agentic_kda_skill"), From 92fe97e6c993be43db9365fb4f1bdffb40be9277 Mon Sep 17 00:00:00 2001 From: Roman Rakus Date: Thu, 8 Oct 2026 15:02:16 +0200 Subject: [PATCH 2/2] feat(gooddata-eval): check that gen-ai sends the Document wire names The evaluator now reads the names gen-ai sends after the Publisher rename: the document answer part with document_ref and the saved_document_id and base_document_id keys, the draft_document tool and the document_builder skill. A new strict check, document_wire_names, fails a run when any Report name is still on the wire: the report part type or keys, the draft_report or list_report_layouts tool, or the report_builder skill. It reads every turn, and its failure leads the list and names each old name, so a half-switched chain shows which part still sends the old one. A user context with view.report is rejected before the first request, because gen-ai reads view.document. This evaluator fails against a gen-ai without the Document names, so release it only once gen-ai sends them. jira: LX-3248 risk: high Co-Authored-By: Claude Opus 5.5 (1M context) --- .../core/agentic/document_skill.py | 115 +++++++++-- .../src/gooddata_eval/core/chat/sse_client.py | 2 +- .../tests/test_agentic_document_skill.py | 185 +++++++++++++++--- .../gooddata-eval/tests/test_chat_render.py | 22 ++- 4 files changed, 266 insertions(+), 58 deletions(-) diff --git a/packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py b/packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py index fcd58ad12..9112f3e02 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/agentic/document_skill.py @@ -49,14 +49,22 @@ # absorb a turn that answered without drafting. _DEFAULT_MAX_ITERATIONS = 4 -_DRAFT_TOOL = "draft_report" -_BUILDER_SKILL = "report_builder" -_PART_TYPE = "report" +_DRAFT_TOOL = "draft_document" +_BUILDER_SKILL = "document_builder" +_PART_TYPE = "document" +_SET_SKILLS_TOOL = "set_skills" +# The Report names gen-ai sent before Publisher's Document names. Any of them on the wire means +# some part of the chain was not switched over. +_LEGACY_PART_TYPE = "report" +_LEGACY_PART_KEYS = ("report", "report_ref", "base_report_id", "saved_report_id") +_LEGACY_DRAFT_TOOL = "draft_report" +_LEGACY_TOOLS = frozenset({_LEGACY_DRAFT_TOOL, "list_report_layouts"}) +_LEGACY_SKILL = "report_builder" # The copilot's own rule: a document opens with a cover and has at least one content page. _COVER_PAGE = "cover" _CONTENT_PAGE = "content" _EXPECTATION_KEYS = frozenset({"period", "visualizations", "narrative", "expects_clarification"}) -# Template tokens the document renders at export time ({reportName}, {periodStart}, ...). +# Template tokens the document renders at export time ({documentName}, {periodStart}, ...). _PLACEHOLDER_RE = re.compile(r"\{\w+\}") _SUMMARY_SLOT = "summary" @@ -82,6 +90,36 @@ def _extract_document_part(chat_result: ChatResult) -> dict | None: return None +def legacy_wire_names(parts: list[dict], tool_call_events: list[ToolCallEvent]) -> list[str]: + """Every Report name the run saw on the wire, in the order it saw them, each named once.""" + found: list[str] = [] + for tc in tool_call_events: + if tc.function_name in _LEGACY_TOOLS: + found.append(f"tool {tc.function_name!r}") + elif tc.function_name == _SET_SKILLS_TOOL: + # The arguments too: a refused call naming the old skill still shows the prompt names it. + arguments = tc.parsed_arguments() + requested = arguments.get("skill_names") if isinstance(arguments, dict) else None + result_data = tc.parsed_result() + payload = result_data.get("data", result_data) if isinstance(result_data, dict) else None + activated = payload.get("skills_to_activate") if isinstance(payload, dict) else None + if any(isinstance(skills, list) and _LEGACY_SKILL in skills for skills in (requested, activated)): + found.append(f"skill {_LEGACY_SKILL!r}") + for part in parts: + if not isinstance(part, dict): + continue + if part.get("type") == _LEGACY_PART_TYPE: + found.append(f"part type {_LEGACY_PART_TYPE!r}") + if part.get("type") not in (_PART_TYPE, _LEGACY_PART_TYPE): + continue + found.extend(f"part key {key!r}" for key in _LEGACY_PART_KEYS if key in part) + for key in (_PART_TYPE, _LEGACY_PART_TYPE): + document = part.get(key) + if isinstance(document, dict) and document.get("type") == _LEGACY_PART_TYPE: + found.append(f"document type {_LEGACY_PART_TYPE!r}") + return list(dict.fromkeys(found)) + + def _nodes(node: Any) -> Iterator[dict]: """Every node of a page layout in document order, following ``row`` and ``column`` as gen-ai does.""" if not isinstance(node, dict): @@ -214,6 +252,19 @@ def _validate_expectation(expected_output: Any) -> None: ) +def _validate_user_context(user_context: dict | None) -> None: + """Reject a user context that names the open document the Report way, before the first API call. + + gen-ai reads ``view.document``; a fixture still sending ``view.report`` would test the old input. + + Raises: + ValueError: the user context carries ``view.report``. + """ + view = (user_context or {}).get("view") + if isinstance(view, dict) and _LEGACY_PART_TYPE in view: + raise ValueError("user_context.view.report is the Report name; send view.document") + + def build_simulated_reply(expected_output: dict) -> str: """The reply the simulated user sends when the copilot asks back instead of drafting. @@ -256,6 +307,7 @@ class DocumentEvaluation: not_saved: bool skill_activated: bool applies: _Applies + wire_names_current: bool = True period_correct: bool = False charts_matched: bool = False summaries_present: bool = False @@ -290,6 +342,7 @@ def strict_checks(self) -> dict[str, bool]: "document_pages_consistent": self.pages_consistent, "document_not_saved": self.not_saved, "document_skill_activated": self.skill_activated, + "document_wire_names": self.wire_names_current, } if self.applies.period: checks["document_period_correct"] = self.period_correct @@ -306,11 +359,11 @@ def _read_document(document_part: dict | None) -> tuple[dict | None, str | None] """The part's document, or why it carries no usable one.""" if document_part is None: return None, f"the response carries no {_PART_TYPE!r} part" - document = document_part.get("report") + document = document_part.get(_PART_TYPE) if not isinstance(document, dict): return ( None, - f"the {_PART_TYPE!r} part carries no document (report_ref {document_part.get('report_ref')!r})", + f"the {_PART_TYPE!r} part carries no document (document_ref {document_part.get('document_ref')!r})", ) if document.get("type") != _PART_TYPE: return None, ( @@ -325,12 +378,31 @@ def evaluate_document_response( expected_output: dict, *, skill_activated: bool, + legacy_names: list[str] | None = None, ) -> DocumentEvaluation: """Score one document response against its expectation. + ``legacy_names`` are the Report names the run saw on the wire, from ``legacy_wire_names``. + Any of them fails ``document_wire_names`` and leads the failures, because a gen-ai still on + the Report names also fails every check that looks for the Document ones. + Pure: no network and no conversation state, so the whole assertion surface is unit-testable without an agent. """ + evaluation = _score_response(tool_result, document_part, expected_output, skill_activated=skill_activated) + if legacy_names: + evaluation.wire_names_current = False + evaluation.failures.insert(0, f"gen-ai sends the Report names: {', '.join(legacy_names)}") + return evaluation + + +def _score_response( + tool_result: dict | None, + document_part: dict | None, + expected_output: dict, + *, + skill_activated: bool, +) -> DocumentEvaluation: applies = _Applies( period=_has_period(expected_output), charts=_has_visualizations(expected_output), @@ -364,7 +436,7 @@ def evaluate_document_response( failures: list[str] = [] - part_ref, tool_ref = document_part.get("report_ref"), tool_result.get("ref") + part_ref, tool_ref = document_part.get("document_ref"), tool_result.get("ref") ref_matches = isinstance(part_ref, str) and bool(part_ref) and part_ref == tool_ref if not ref_matches: failures.append(f"the {_PART_TYPE!r} part shows {part_ref!r}, but {_DRAFT_TOOL} returned {tool_ref!r}") @@ -385,7 +457,7 @@ def evaluate_document_response( pages_consistent = bool(kinds) and kinds[0] == _COVER_PAGE and _CONTENT_PAGE in kinds not_saved = True - for key, wording in (("saved_report_id", "be saved yet"), ("base_report_id", "edit a saved document")): + for key, wording in (("saved_document_id", "be saved yet"), ("base_document_id", "edit a saved document")): value = document_part.get(key) if value is not None: failures.append(f"a new draft must not {wording}, but it reports {key} {value!r}") @@ -442,7 +514,7 @@ def _judge_narrative( No document means nothing to grade, which is a failed narrative rather than a judge call. """ - document = (document_part or {}).get("report") if evaluation.part_present else None + document = (document_part or {}).get(_PART_TYPE) if evaluation.part_present else None if not isinstance(document, dict): return 0.0 started = time.monotonic() @@ -542,6 +614,7 @@ def _execute_single_document_run( response_id: str | None = None all_tool_call_events: list[ToolCallEvent] = [] all_reasoning_step_events: list[ReasoningStepEvent] = [] + all_parts: list[dict] = [] timings = PhaseTimings() turn_offset = 0.0 tool_index_offset = 0 @@ -564,6 +637,7 @@ def _execute_single_document_run( ) all_tool_call_events.extend(chat_result.tool_call_events or []) all_reasoning_step_events.extend(chat_result.reasoning_step_events or []) + all_parts.extend(chat_result.unhandled_parts) steps += chat_result.reasoning_step_count candidate = _extract_tool_result(chat_result.tool_call_events or [], _DRAFT_TOOL) @@ -576,6 +650,9 @@ def _execute_single_document_run( # Read from the same turn as the draft: the part shows the version that call stored. document_part = _extract_document_part(chat_result) break + if _extract_tool_result(chat_result.tool_call_events or [], _LEGACY_DRAFT_TOOL) is not None: + # A gen-ai on the Report names drafted under the old tool; another reply cannot change that. + break response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result) if iteration == 0: @@ -595,6 +672,7 @@ def _execute_single_document_run( document_part, expected_output, skill_activated=_skill_activated(all_tool_call_events, _BUILDER_SKILL), + legacy_names=legacy_wire_names(all_parts, all_tool_call_events), ) if judge is not None and _has_narrative(expected_output): timings.judge_s += _judge_narrative(evaluation, judge, document_part, question, expected_output["narrative"]) @@ -636,9 +714,11 @@ def run_agentic_document_skill( without one needs neither the llm-judge extra nor ``OPENAI_API_KEY``. Raises: - ValueError: the fixture is unusable — see ``_validate_expectation``. + ValueError: the fixture is unusable — see ``_validate_expectation`` and + ``_validate_user_context``. """ _validate_expectation(expected_output) + _validate_user_context(user_context) if judge is None and _has_narrative(expected_output): judge = LLMJudge(_NARRATIVE_EVALUATION_STEPS) run_results: list[DocumentRunResult] = [] @@ -716,7 +796,8 @@ def evaluate_agentic_document_skill( Raises: DocumentSkillAssertionError: the gate did not pass. - ValueError: the fixture is unusable — see ``_validate_expectation``. Raised before any + ValueError: the fixture is unusable — see ``_validate_expectation`` and + ``_validate_user_context``. Raised before any request, so it means a fixture to fix rather than a result to read. JudgeResponseError: the fixture states a narrative and the judge returned no readable verdict for any run -- an item without a result, not K failures. @@ -832,15 +913,15 @@ def _run_detail(run: DocumentRunResult) -> dict[str, Any]: if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k): gate_note = gate_failure_note(gate, runs_passed, runs_effective, len(unscored)) - skill_note = ( - "" - if best.evaluation.skill_activated - else ( + skill_note = "" + if not best.evaluation.wire_names_current: + skill_note = " gen-ai still sends the Report names, so it predates the Document rename." + elif not best.evaluation.skill_activated: + skill_note = ( f" set_skills never activated {_BUILDER_SKILL}: either the copilot routed elsewhere, or the skill" - " is not registered. It registers only with enableGenAiReportBuilderSkill and the org's" + " is not registered. It registers only with enableGenAiDocumentBuilderSkill and the org's" " enableBusinessBriefingReportsApp both on." ) - ) exc = DocumentSkillAssertionError( f"Document skill assertion failed. {gate_note}{skill_note} " f"Checks: {best.evaluation.strict_checks}. " diff --git a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py index e086db8ce..ba7dae57b 100644 --- a/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py +++ b/packages/gooddata-eval/src/gooddata_eval/core/chat/sse_client.py @@ -53,7 +53,7 @@ "visualization", "dashboard", "dashboardPatch", - "report", + "document", "kda", "whatIf", "searchResults", diff --git a/packages/gooddata-eval/tests/test_agentic_document_skill.py b/packages/gooddata-eval/tests/test_agentic_document_skill.py index 75f4d84b9..baa60b1de 100644 --- a/packages/gooddata-eval/tests/test_agentic_document_skill.py +++ b/packages/gooddata-eval/tests/test_agentic_document_skill.py @@ -19,6 +19,7 @@ build_simulated_reply, evaluate_agentic_document_skill, evaluate_document_response, + legacy_wire_names, run_agentic_document_skill, ) from gooddata_eval.core.evaluators._llm_judge import JudgeResponseError @@ -36,7 +37,7 @@ def _cover_page() -> dict: "id": "page1", "kind": "cover", "format": "widescreen", - "layout": {"column": [{"id": "coverTitle", "weight": 2, "heading": "{reportName}", "style": "h1"}]}, + "layout": {"column": [{"id": "coverTitle", "weight": 2, "heading": "{documentName}", "style": "h1"}]}, } @@ -66,15 +67,15 @@ def _document_part( ref: str = "document_1", page_count: int | None = None, period: dict | None = None, - saved_report_id: str | None = None, - base_report_id: str | None = None, + saved_document_id: str | None = None, + base_document_id: str | None = None, document: dict | None | str = "default", ) -> dict: pages = pages if pages is not None else [_cover_page(), _content_page("page2", _REVENUE_TREND)] document = ( { "id": "sales_overview", - "type": "report", + "type": "document", "title": "Sales overview", "period": period if period is not None else _PERIOD, "pages": pages, @@ -83,13 +84,13 @@ def _document_part( else document ) return { - "type": "report", - "report_ref": ref, + "type": "document", + "document_ref": ref, "format": "aac-v1", - "report": document, + "document": document, "page_count": len(pages) if page_count is None else page_count, - "base_report_id": base_report_id, - "saved_report_id": saved_report_id, + "base_document_id": base_document_id, + "saved_document_id": saved_document_id, } @@ -130,7 +131,7 @@ def _chat_result( def _drafting_turn(**part_kwargs: Any) -> ChatResult: return _chat_result( - tool_calls=[_set_skills("report_builder"), _tool_call("draft_report", _draft_result())], + tool_calls=[_set_skills("document_builder"), _tool_call("draft_document", _draft_result())], parts=[{"type": "text", "text": "I've put together a document with 2 slides."}, _document_part(**part_kwargs)], text="I've put together a document with 2 slides.", ) @@ -189,6 +190,7 @@ def test_a_drafted_document_returned_in_the_chat_item_passes() -> None: "document_pages_consistent": True, "document_not_saved": True, "document_skill_activated": True, + "document_wire_names": True, } assert evaluation.strict_pass assert evaluation.failures == [] @@ -207,38 +209,38 @@ def test_no_successful_draft_fails_every_check_and_says_so() -> None: assert evaluation.strict_checks["document_part_present"] is False assert evaluation.strict_checks["document_period_correct"] is False assert not evaluation.strict_pass - assert evaluation.failures == ["the agent never produced a successful draft_report call"] + assert evaluation.failures == ["the agent never produced a successful draft_document call"] def test_a_draft_without_a_document_part_fails() -> None: evaluation = _evaluate(None) assert evaluation.strict_checks["document_drafted"] is True assert evaluation.strict_checks["document_part_present"] is False - assert evaluation.failures == ["the response carries no 'report' part"] + assert evaluation.failures == ["the response carries no 'document' part"] def test_a_document_part_whose_document_did_not_resolve_fails() -> None: evaluation = _evaluate(_document_part(document=None)) assert evaluation.strict_checks["document_part_present"] is False - assert evaluation.failures == ["the 'report' part carries no document (report_ref 'document_1')"] + assert evaluation.failures == ["the 'document' part carries no document (document_ref 'document_1')"] def test_a_document_that_is_not_a_document_fails() -> None: evaluation = _evaluate(_document_part(document={"id": "x", "type": "dashboard", "pages": []})) assert evaluation.strict_checks["document_part_present"] is False - assert evaluation.failures == ["the 'report' part carries a document of type 'dashboard', expected 'report'"] + assert evaluation.failures == ["the 'document' part carries a document of type 'dashboard', expected 'document'"] def test_a_part_pointing_at_another_draft_fails() -> None: evaluation = _evaluate(_document_part(ref="document_2")) assert evaluation.strict_checks["document_ref_matches"] is False - assert evaluation.failures == ["the 'report' part shows 'document_2', but draft_report returned 'document_1'"] + assert evaluation.failures == ["the 'document' part shows 'document_2', but draft_document returned 'document_1'"] def test_a_page_count_that_disagrees_with_the_pages_fails() -> None: evaluation = _evaluate(_document_part(page_count=3)) assert evaluation.strict_checks["document_pages_consistent"] is False - assert evaluation.failures == ["the document has 2 page(s), but the part says 3 and draft_report said 2"] + assert evaluation.failures == ["the document has 2 page(s), but the part says 3 and draft_document said 2"] def test_a_cover_alone_is_not_a_document() -> None: @@ -248,16 +250,18 @@ def test_a_cover_alone_is_not_a_document() -> None: def test_a_new_draft_must_not_already_be_saved() -> None: - evaluation = _evaluate(_document_part(saved_report_id="sales_overview")) + evaluation = _evaluate(_document_part(saved_document_id="sales_overview")) assert evaluation.strict_checks["document_not_saved"] is False - assert evaluation.failures == ["a new draft must not be saved yet, but it reports saved_report_id 'sales_overview'"] + assert evaluation.failures == [ + "a new draft must not be saved yet, but it reports saved_document_id 'sales_overview'" + ] def test_a_new_draft_must_not_edit_a_saved_document() -> None: - evaluation = _evaluate(_document_part(base_report_id="q3_review")) + evaluation = _evaluate(_document_part(base_document_id="q3_review")) assert evaluation.strict_checks["document_not_saved"] is False assert evaluation.failures == [ - "a new draft must not edit a saved document, but it reports base_report_id 'q3_review'" + "a new draft must not edit a saved document, but it reports base_document_id 'q3_review'" ] @@ -289,6 +293,126 @@ def test_visualizations_are_found_however_deep_the_layout_nests_them() -> None: assert _visualizations_of({"pages": [_cover_page(), page]}) == {_REVENUE_TREND, _RETURNS_BY_CATEGORY} +def _legacy_part() -> dict: + """The answer part as gen-ai sent it before the Document names.""" + part = _document_part() + document = {**part["document"], "type": "report"} + return { + "type": "report", + "report_ref": part["document_ref"], + "format": "aac-v1", + "report": document, + "page_count": part["page_count"], + "base_report_id": None, + "saved_report_id": None, + } + + +def test_a_part_with_the_report_names_fails_the_wire_check_and_names_each_one() -> None: + names = legacy_wire_names([_legacy_part()], []) + assert names == [ + "part type 'report'", + "part key 'report'", + "part key 'report_ref'", + "part key 'base_report_id'", + "part key 'saved_report_id'", + "document type 'report'", + ] + + +def test_the_report_tool_and_skill_ids_are_named_too() -> None: + turn = _chat_result( + tool_calls=[ + _set_skills("report_builder"), + _tool_call("list_report_layouts", {"status": "success"}), + _tool_call("draft_report", _draft_result()), + ] + ) + assert legacy_wire_names([], turn.tool_call_events) == [ + "skill 'report_builder'", + "tool 'list_report_layouts'", + "tool 'draft_report'", + ] + + +def test_a_set_skills_call_asking_for_the_report_skill_is_named_even_when_it_was_refused() -> None: + call = { + "functionName": "set_skills", + "functionArguments": json.dumps({"skill_names": ["report_builder"]}), + "result": json.dumps({"status": "error", "message": "unknown skill"}), + } + turn = _chat_result(tool_calls=[call]) + assert legacy_wire_names([], turn.tool_call_events) == ["skill 'report_builder'"] + + +def test_a_set_skills_call_whose_arguments_are_not_an_object_is_ignored() -> None: + call = {"functionName": "set_skills", "functionArguments": json.dumps(["report_builder"]), "result": None} + assert legacy_wire_names([], _chat_result(tool_calls=[call]).tool_call_events) == [] + + +def test_the_document_names_leave_the_wire_check_passing() -> None: + turn = _drafting_turn() + assert legacy_wire_names(turn.unhandled_parts, turn.tool_call_events) == [] + + +def test_a_run_against_a_gen_ai_with_the_report_names_fails_on_the_wire_check() -> None: + turn = _chat_result( + tool_calls=[_set_skills("report_builder"), _tool_call("draft_report", _draft_result())], + parts=[_legacy_part()], + ) + run = _execute_single_document_run(_ScriptedChatClient([turn]), "conv-1", "Create a document", {}, 1) + + assert run.evaluation.strict_checks["document_wire_names"] is False + assert run.evaluation.failures[0] == ( + "gen-ai sends the Report names: skill 'report_builder', tool 'draft_report', part type 'report', " + "part key 'report', part key 'report_ref', part key 'base_report_id', part key 'saved_report_id', " + "document type 'report'" + ) + + +def test_the_wire_check_reads_every_turn_not_only_the_drafting_one() -> None: + asking = _chat_result(parts=[{"type": "report", "report_ref": "report_0"}], text="Which period?") + run = _execute_single_document_run( + _ScriptedChatClient([asking, _drafting_turn()]), "conv-1", "Create a document", {}, 2 + ) + + assert run.evaluation.strict_checks["document_wire_names"] is False + + +def test_a_gen_ai_on_the_report_names_stops_at_its_draft_instead_of_replying_again() -> None: + turn = _chat_result( + tool_calls=[_set_skills("report_builder"), _tool_call("draft_report", _draft_result())], + parts=[_legacy_part()], + ) + client = _ScriptedChatClient([turn, _drafting_turn()]) + run = _execute_single_document_run(client, "conv-1", "Create a document", {}, 4) + + assert client.sent == ["Create a document"] + assert run.evaluation.strict_checks["document_wire_names"] is False + + +def test_a_failing_item_on_the_report_names_points_at_the_rename_not_the_flags( + monkeypatch: pytest.MonkeyPatch, +) -> None: + turn = _chat_result(tool_calls=[_set_skills("report_builder"), _tool_call("draft_report", _draft_result())]) + _install_client(monkeypatch, _ScriptedChatClient([turn])) + with pytest.raises(DocumentSkillAssertionError) as raised: + evaluate_agentic_document_skill("https://h", "tok", "ws", "Create a document", {}, max_iterations=1) + + assert "gen-ai still sends the Report names" in str(raised.value) + assert "enableGenAiDocumentBuilderSkill" not in str(raised.value) + + +def test_a_user_context_with_view_report_fails_before_any_request(monkeypatch: pytest.MonkeyPatch) -> None: + client = _ScriptedChatClient([_drafting_turn()]) + _install_client(monkeypatch, client) + with pytest.raises(ValueError, match="user_context.view.report is the Report name; send view.document"): + run_agentic_document_skill( + "https://h", "tok", "ws", "Refine it", {}, user_context={"view": {"report": {"id": "q3"}}} + ) + assert client.sent == [] + + # ── fixture validation ────────────────────────────────────────────────────── @@ -412,17 +536,17 @@ def test_a_failing_item_raises_with_the_failures(monkeypatch: pytest.MonkeyPatch _install_client(monkeypatch, _ScriptedChatClient([_chat_result(text="I can't do that.")])) with pytest.raises(DocumentSkillAssertionError) as raised: evaluate_agentic_document_skill("https://h", "tok", "ws", "Create a document", {}, max_iterations=1) - assert "the agent never produced a successful draft_report call" in str(raised.value) + assert "the agent never produced a successful draft_document call" in str(raised.value) assert raised.value.runs_passed == 0 assert raised.value.conversation_id == "conv-1" -def test_a_failing_item_without_report_builder_points_at_the_flags(monkeypatch: pytest.MonkeyPatch) -> None: +def test_a_failing_item_without_document_builder_points_at_the_flags(monkeypatch: pytest.MonkeyPatch) -> None: turn = _chat_result(tool_calls=[_set_skills("visualization")], text="Here is a chart.") _install_client(monkeypatch, _ScriptedChatClient([turn])) with pytest.raises( DocumentSkillAssertionError, - match="enableGenAiReportBuilderSkill and the org.s enableBusinessBriefingReportsApp", + match="enableGenAiDocumentBuilderSkill and the org.s enableBusinessBriefingReportsApp", ): evaluate_agentic_document_skill("https://h", "tok", "ws", "Create a document", {}, max_iterations=1) @@ -443,9 +567,9 @@ def test_a_ref_missing_on_both_sides_does_not_match() -> None: def test_the_last_successful_draft_of_a_turn_is_the_one_scored() -> None: turn = _chat_result( tool_calls=[ - _tool_call("draft_report", _draft_result(ref="document_1")), - _tool_call("draft_report", {"status": "error", "message": "no layout fits 7 charts"}), - _tool_call("draft_report", _draft_result(ref="document_2")), + _tool_call("draft_document", _draft_result(ref="document_1")), + _tool_call("draft_document", {"status": "error", "message": "no layout fits 7 charts"}), + _tool_call("draft_document", _draft_result(ref="document_2")), ], parts=[_document_part(ref="document_2")], text="I've put together a document with 2 slides.", @@ -541,13 +665,13 @@ def test_asking_first_is_recorded_whenever_the_fixture_states_the_key() -> None: def test_the_last_document_part_of_a_turn_is_the_one_scored() -> None: turn = _chat_result( tool_calls=[ - _tool_call("draft_report", _draft_result(ref="document_1")), - _tool_call("draft_report", _draft_result(ref="document_2")), + _tool_call("draft_document", _draft_result(ref="document_1")), + _tool_call("draft_document", _draft_result(ref="document_2")), ], parts=[_document_part(ref="document_1"), _document_part(ref="document_2")], ) run = _execute_single_document_run(_ScriptedChatClient([turn]), "c", "q", {}, max_iterations=4) - assert run.document_part is not None and run.document_part["report_ref"] == "document_2" + assert run.document_part is not None and run.document_part["document_ref"] == "document_2" assert run.evaluation.strict_pass @@ -607,6 +731,7 @@ def test_every_scored_check_reaches_langfuse_with_the_draft_facts(monkeypatch: p "document_pages_consistent", "document_not_saved", "document_skill_activated", + "document_wire_names", "document_period_correct", "document_asked_first", "pass_at_k", @@ -657,7 +782,7 @@ def score(self, input: str, expected_output: str, actual_output: str) -> tuple[b def _narrative_turn(*pages: dict) -> ChatResult: return _chat_result( - tool_calls=[_tool_call("draft_report", _draft_result(page_count=len(pages)))], + tool_calls=[_tool_call("draft_document", _draft_result(page_count=len(pages)))], parts=[_document_part(list(pages))], text="I've put together a document.", ) diff --git a/packages/gooddata-eval/tests/test_chat_render.py b/packages/gooddata-eval/tests/test_chat_render.py index 1baad5a6b..29904158f 100644 --- a/packages/gooddata-eval/tests/test_chat_render.py +++ b/packages/gooddata-eval/tests/test_chat_render.py @@ -52,14 +52,14 @@ def test_an_unmodelled_part_type_is_kept_not_dropped(): assert "kda" in render_answer_text(result) -def test_a_report_part_is_kept_verbatim_without_an_unknown_type_warning(caplog: pytest.LogCaptureFixture) -> None: +def test_a_document_part_is_kept_verbatim_without_an_unknown_type_warning(caplog: pytest.LogCaptureFixture) -> None: part = { - "type": "report", - "report_ref": "report_1", + "type": "document", + "document_ref": "document_1", "format": "aac-v1", - "report": { + "document": { "id": "sales_overview", - "type": "report", + "type": "document", "title": "Sales overview", "period": {"start": "2026-01-01", "end": "2026-06-30"}, "pages": [ @@ -67,7 +67,9 @@ def test_a_report_part_is_kept_verbatim_without_an_unknown_type_warning(caplog: "id": "page1", "kind": "cover", "format": "widescreen", - "layout": {"column": [{"id": "coverTitle", "weight": 2, "heading": "{reportName}", "style": "h1"}]}, + "layout": { + "column": [{"id": "coverTitle", "weight": 2, "heading": "{documentName}", "style": "h1"}] + }, }, { "id": "page2", @@ -83,11 +85,11 @@ def test_a_report_part_is_kept_verbatim_without_an_unknown_type_warning(caplog: ], }, "page_count": 2, - "base_report_id": None, - "saved_report_id": None, + "base_document_id": None, + "saved_document_id": None, } with caplog.at_level("WARNING", logger="gooddata_eval.core.chat.sse_client"): - result = parse_sse_lines(_multipart_lines({"type": "text", "text": "I've put together a report."}, part)) + result = parse_sse_lines(_multipart_lines({"type": "text", "text": "I've put together a document."}, part)) assert result.unhandled_parts == [part] assert "unknown multipart part type" not in caplog.text @@ -110,7 +112,7 @@ def test_known_part_types_matches_the_documented_gen_ai_union(): "visualization", "dashboard", "dashboardPatch", - "report", + "document", "kda", "whatIf", "searchResults",