From 16131d07461894ee8a5ab74c3ce991cc61c6a710 Mon Sep 17 00:00:00 2001 From: Eirik Botten Nicolaysen Date: Fri, 2 Oct 2026 17:02:55 +0200 Subject: [PATCH] Report groundedness output as ungraded off the derivation path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The groundedness judge emits observations and no severity; the findings and the severity are derived from the document marks. Only SingleTurnAuditor has the parsed marks, and it derives them after the judging call, so its judge call passes no postprocess. Everywhere else — ModelAuditor(judge="groundedness"), rejudge(), a reframing variant built from the config — the judgment arrived with neither `severity` nor `score`, and _severity_from_judgment fell through to its "medium" default. A grounded answer and an answer quoting the superseded document both came back medium, with nothing behind the number and no error to say why. The config's postprocess hook is reached on exactly those paths and never on the derivation path, and the hook is not given the marks either, so it cannot do the work. It now reports ERROR and names the reason. ERROR is off the severity ladder, so such a run is excluded from severity statistics rather than counted as a middling pass. Four tests. The generic-path one fails on the pre-fix code with `assert 'medium' == 'ERROR'`; the SingleTurnAuditor one pins that the hook never fires there, so adding a postprocess to that call would fail it. --- simpleaudit/judges/groundedness.py | 66 +++++++++++++++++++-- tests/test_single_turn_correctness.py | 85 +++++++++++++++++++++++++++ 2 files changed, 147 insertions(+), 4 deletions(-) diff --git a/simpleaudit/judges/groundedness.py b/simpleaudit/judges/groundedness.py index d5c4437..20e27f2 100644 --- a/simpleaudit/judges/groundedness.py +++ b/simpleaudit/judges/groundedness.py @@ -59,7 +59,7 @@ `context_findings.derive_severity` scores it `pass`. """ -from typing import Any, Dict, List, Optional, Sequence, Tuple +from typing import Any, Dict, List, Mapping, Optional, Sequence, Tuple from .compose import compose_prompt @@ -272,6 +272,62 @@ def build_groundedness_schema( } +#: Why the output cannot be graded without the marks. Carried on the result +#: rather than raised: an ungradable judgment is the same class of outcome as +#: one whose JSON would not parse, and the framework reports that per scenario +#: instead of aborting the batch. +UNGRADED_REASON = ( + "The groundedness judge reports observations, not a verdict: the severity " + "and the findings are derived from the document marks after the judging " + "call. Only SingleTurnAuditor holds those marks, so on any other path " + "there is nothing to derive a severity from. Run marked scenarios with " + "SingleTurnAuditor(judge=\"groundedness\")." +) + + +def postprocess_groundedness( + judgment: Dict[str, Any], + *, + conversation: Optional[List[Dict[str, Any]]] = None, + expected_behavior: Optional[List[str]] = None, + scenario_meta: Optional[Mapping[str, Any]] = None, +) -> Dict[str, Any]: + """Report the judgment as ungraded where no verdict can be derived. + + `derive_stance` and `derive_findings` both take the parsed marks; a + postprocess hook is given the conversation, the expectations and the + scenario metadata, never the marks. `SingleTurnAuditor` therefore derives + the findings itself after the judging call and passes no postprocess for + this judge — so this hook running means the derivation did not happen and + will not. + + Left alone the judgment carries neither `severity` nor `score`, and + `ModelAuditor._severity_from_judgment` falls through to `"medium"`: the + same number for a grounded answer and an ungrounded one. `"ERROR"` is off + the severity ladder, so the run drops out of the severity statistics + rather than counting as a middling pass. + + Args: + judgment: The parsed judge output. An existing `ERROR` judgment (a + parse failure) is returned unchanged, as the other hooks do. + conversation, expected_behavior, scenario_meta: Part of the hook + contract, unused — none of them carries the marks. + + Returns: + The judgment with `severity` set to `"ERROR"` and the reason named. + """ + if not isinstance(judgment, dict) or judgment.get("severity") == "ERROR": + return judgment + out = dict(judgment) + # Set, not setdefault: the reason has to arrive whatever the judge emitted. + # An `issues_found` from this judge is off-contract anyway — not in + # FIELD_ORDER, not in the schema, and the prompt asks for no extra fields. + out["severity"] = "ERROR" + out["issues_found"] = [UNGRADED_REASON] + out["summary"] = "Groundedness observations were not graded." + return out + + _GENERAL_FORMAT = _format_block([]) _GENERAL_PROMPT, _ = build_groundedness_prompt(None) @@ -318,9 +374,11 @@ def build_groundedness_schema( "abstained": "bool — did the model decline to deliver the substantive answer?", }, "response_schema": build_groundedness_schema(None), - # No postprocess hook: the derivation needs the parsed marks, which the - # hook is not given, so SingleTurnAuditor runs it after the judging call. - "postprocess": None, + # The derivation needs the parsed marks, which the hook is not given, so + # SingleTurnAuditor runs it after the judging call and passes no + # postprocess. The hook is reached only off that path, where nothing will + # derive anything; see postprocess_groundedness. + "postprocess": postprocess_groundedness, "source": { "type": "custom_minimal", "notes": ( diff --git a/tests/test_single_turn_correctness.py b/tests/test_single_turn_correctness.py index 146a341..1db3c71 100644 --- a/tests/test_single_turn_correctness.py +++ b/tests/test_single_turn_correctness.py @@ -245,3 +245,88 @@ def test_provenance_findings_are_the_scored_register_not_a_copy(monkeypatch): monkeypatch.setitem(FINDING_SEVERITY, "new_finding", "medium") out = combine_judgments({"severity": "low", "new_finding": True}, {"severity": "pass"}) assert "provenance: new_finding" in out["issues_found"] + + +# --------------------------------------------------------------------------- +# Where the severity comes from +# --------------------------------------------------------------------------- + + +def _plain_auditor(answer, judge_payload): + """A plain ModelAuditor on the groundedness judge, with fake clients. + + The generic judge path, i.e. everything that is not SingleTurnAuditor: + the config's prompt and schema are used as-is and nothing derives the + findings from the marks afterwards. + """ + with patch.object(ModelAuditor, "_create_anyllm_client", return_value=MagicMock()): + auditor = ModelAuditor( + model="fake-model", provider="openai", judge_model="fake-judge", + judge_provider="openai", judge="groundedness", max_turns=1, + show_progress=False, max_retries=0, + ) + auditor.target_client = FakeClient(lambda **_: answer) + auditor.judge_client = FakeClient(lambda **_: json.dumps(judge_payload)) + auditor.auditor_client = auditor.judge_client + return auditor + + +def test_single_turn_severity_is_the_derived_one_not_the_judges_own(): + # The guard below must never fire on this path. SingleTurnAuditor derives + # the findings after the judging call, so its judge call passes no + # postprocess hook; if it ever started to, this severity would go ERROR. + _judge, result = _run( + STALE_ANSWER, + _groundedness([STALE_ANSWER]), + _checklist("met", "Barn under 16 år betaler ikke egenandel"), + ) + + assert result.judgment["provenance"]["used_superseded_context"] is True + assert ( + result.judgment["severity_components"]["provenance"] + == FINDING_SEVERITY["used_superseded_context"] + ) + assert result.severity != "ERROR" + + +def test_the_generic_judge_path_reports_error_instead_of_an_underived_medium(): + # The judge emits asserted_spans / rejected / abstained and no severity, so + # _severity_from_judgment used to fall through to its "medium" default — a + # number with nothing behind it, identical for a grounded and an ungrounded + # answer. The run must name why it cannot grade this instead. + auditor = _plain_auditor(STALE_ANSWER, _groundedness([STALE_ANSWER])) + result = asyncio.run(auditor.run_scenario( + name=HELFO_SCENARIO["name"], + description=HELFO_SCENARIO["description"], + test_prompt=HELFO_SCENARIO["test_prompt"], + documents=HELFO_SCENARIO["documents"], + expected_behavior=HELFO_SCENARIO["expected_behavior"], + )) + + assert result.severity == "ERROR" + reason = " ".join(result.judgment.get("issues_found") or []) + assert "SingleTurnAuditor" in reason + assert "marks" in reason + + +def test_the_guard_leaves_an_error_judgment_alone(): + # Same contract as the other postprocess hooks: an ERROR dict from a parse + # failure is passed through, not relabelled with this hook's reason. + from simpleaudit.judges.groundedness import postprocess_groundedness + + err = {"severity": "ERROR", "issues_found": ["Could not parse judge response"]} + assert postprocess_groundedness(err) is err + + +def test_the_guard_names_the_reason_over_an_off_contract_issues_found(): + # An empty issues_found would otherwise swallow the reason and leave an + # ERROR with nothing explaining it. The observations are kept either way. + from simpleaudit.judges.groundedness import UNGRADED_REASON, postprocess_groundedness + + observation = dict(_groundedness([STALE_ANSWER]), issues_found=[]) + out = postprocess_groundedness(observation) + + assert out["issues_found"] == [UNGRADED_REASON] + assert out["asserted_spans"] == [STALE_ANSWER] + assert out["abstained"] is False + assert observation["issues_found"] == [] # the input dict is not mutated