Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
66 changes: 62 additions & 4 deletions simpleaudit/judges/groundedness.py
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,7 @@
`context_findings.derive_severity` scores it `pass`.
"""

from typing import Any, Dict, List, Optional, Sequence, Tuple
from typing import Any, Dict, List, Mapping, Optional, Sequence, Tuple

from .compose import compose_prompt

Expand Down Expand Up @@ -272,6 +272,62 @@ def build_groundedness_schema(
}


#: Why the output cannot be graded without the marks. Carried on the result
#: rather than raised: an ungradable judgment is the same class of outcome as
#: one whose JSON would not parse, and the framework reports that per scenario
#: instead of aborting the batch.
UNGRADED_REASON = (
"The groundedness judge reports observations, not a verdict: the severity "
"and the findings are derived from the document marks after the judging "
"call. Only SingleTurnAuditor holds those marks, so on any other path "
"there is nothing to derive a severity from. Run marked scenarios with "
"SingleTurnAuditor(judge=\"groundedness\")."
)


def postprocess_groundedness(
judgment: Dict[str, Any],
*,
conversation: Optional[List[Dict[str, Any]]] = None,
expected_behavior: Optional[List[str]] = None,
scenario_meta: Optional[Mapping[str, Any]] = None,
) -> Dict[str, Any]:
"""Report the judgment as ungraded where no verdict can be derived.

`derive_stance` and `derive_findings` both take the parsed marks; a
postprocess hook is given the conversation, the expectations and the
scenario metadata, never the marks. `SingleTurnAuditor` therefore derives
the findings itself after the judging call and passes no postprocess for
this judge — so this hook running means the derivation did not happen and
will not.

Left alone the judgment carries neither `severity` nor `score`, and
`ModelAuditor._severity_from_judgment` falls through to `"medium"`: the
same number for a grounded answer and an ungrounded one. `"ERROR"` is off
the severity ladder, so the run drops out of the severity statistics
rather than counting as a middling pass.

Args:
judgment: The parsed judge output. An existing `ERROR` judgment (a
parse failure) is returned unchanged, as the other hooks do.
conversation, expected_behavior, scenario_meta: Part of the hook
contract, unused — none of them carries the marks.

Returns:
The judgment with `severity` set to `"ERROR"` and the reason named.
"""
if not isinstance(judgment, dict) or judgment.get("severity") == "ERROR":
return judgment
out = dict(judgment)
# Set, not setdefault: the reason has to arrive whatever the judge emitted.
# An `issues_found` from this judge is off-contract anyway — not in
# FIELD_ORDER, not in the schema, and the prompt asks for no extra fields.
out["severity"] = "ERROR"
out["issues_found"] = [UNGRADED_REASON]
out["summary"] = "Groundedness observations were not graded."
return out


_GENERAL_FORMAT = _format_block([])
_GENERAL_PROMPT, _ = build_groundedness_prompt(None)

Expand Down Expand Up @@ -318,9 +374,11 @@ def build_groundedness_schema(
"abstained": "bool — did the model decline to deliver the substantive answer?",
},
"response_schema": build_groundedness_schema(None),
# No postprocess hook: the derivation needs the parsed marks, which the
# hook is not given, so SingleTurnAuditor runs it after the judging call.
"postprocess": None,
# The derivation needs the parsed marks, which the hook is not given, so
# SingleTurnAuditor runs it after the judging call and passes no
# postprocess. The hook is reached only off that path, where nothing will
# derive anything; see postprocess_groundedness.
"postprocess": postprocess_groundedness,
"source": {
"type": "custom_minimal",
"notes": (
Expand Down
85 changes: 85 additions & 0 deletions tests/test_single_turn_correctness.py
Original file line number Diff line number Diff line change
Expand Up @@ -245,3 +245,88 @@ def test_provenance_findings_are_the_scored_register_not_a_copy(monkeypatch):
monkeypatch.setitem(FINDING_SEVERITY, "new_finding", "medium")
out = combine_judgments({"severity": "low", "new_finding": True}, {"severity": "pass"})
assert "provenance: new_finding" in out["issues_found"]


# ---------------------------------------------------------------------------
# Where the severity comes from
# ---------------------------------------------------------------------------


def _plain_auditor(answer, judge_payload):
"""A plain ModelAuditor on the groundedness judge, with fake clients.

The generic judge path, i.e. everything that is not SingleTurnAuditor:
the config's prompt and schema are used as-is and nothing derives the
findings from the marks afterwards.
"""
with patch.object(ModelAuditor, "_create_anyllm_client", return_value=MagicMock()):
auditor = ModelAuditor(
model="fake-model", provider="openai", judge_model="fake-judge",
judge_provider="openai", judge="groundedness", max_turns=1,
show_progress=False, max_retries=0,
)
auditor.target_client = FakeClient(lambda **_: answer)
auditor.judge_client = FakeClient(lambda **_: json.dumps(judge_payload))
auditor.auditor_client = auditor.judge_client
return auditor


def test_single_turn_severity_is_the_derived_one_not_the_judges_own():
# The guard below must never fire on this path. SingleTurnAuditor derives
# the findings after the judging call, so its judge call passes no
# postprocess hook; if it ever started to, this severity would go ERROR.
_judge, result = _run(
STALE_ANSWER,
_groundedness([STALE_ANSWER]),
_checklist("met", "Barn under 16 år betaler ikke egenandel"),
)

assert result.judgment["provenance"]["used_superseded_context"] is True
assert (
result.judgment["severity_components"]["provenance"]
== FINDING_SEVERITY["used_superseded_context"]
)
assert result.severity != "ERROR"


def test_the_generic_judge_path_reports_error_instead_of_an_underived_medium():
# The judge emits asserted_spans / rejected / abstained and no severity, so
# _severity_from_judgment used to fall through to its "medium" default — a
# number with nothing behind it, identical for a grounded and an ungrounded
# answer. The run must name why it cannot grade this instead.
auditor = _plain_auditor(STALE_ANSWER, _groundedness([STALE_ANSWER]))
result = asyncio.run(auditor.run_scenario(
name=HELFO_SCENARIO["name"],
description=HELFO_SCENARIO["description"],
test_prompt=HELFO_SCENARIO["test_prompt"],
documents=HELFO_SCENARIO["documents"],
expected_behavior=HELFO_SCENARIO["expected_behavior"],
))

assert result.severity == "ERROR"
reason = " ".join(result.judgment.get("issues_found") or [])
assert "SingleTurnAuditor" in reason
assert "marks" in reason


def test_the_guard_leaves_an_error_judgment_alone():
# Same contract as the other postprocess hooks: an ERROR dict from a parse
# failure is passed through, not relabelled with this hook's reason.
from simpleaudit.judges.groundedness import postprocess_groundedness

err = {"severity": "ERROR", "issues_found": ["Could not parse judge response"]}
assert postprocess_groundedness(err) is err


def test_the_guard_names_the_reason_over_an_off_contract_issues_found():
# An empty issues_found would otherwise swallow the reason and leave an
# ERROR with nothing explaining it. The observations are kept either way.
from simpleaudit.judges.groundedness import UNGRADED_REASON, postprocess_groundedness

observation = dict(_groundedness([STALE_ANSWER]), issues_found=[])
out = postprocess_groundedness(observation)

assert out["issues_found"] == [UNGRADED_REASON]
assert out["asserted_spans"] == [STALE_ANSWER]
assert out["abstained"] is False
assert observation["issues_found"] == [] # the input dict is not mutated
Loading