diff --git a/.agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py b/.agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py index 771b2c8..2a5c8ee 100644 --- a/.agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py +++ b/.agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py @@ -462,9 +462,11 @@ def _blinded_report(source): ("p", "summary-provenance"), ("p", "summary-status"), ("div", "summary-pair"), - ("nav", "evidence-nav"), ] ) + if any(node.has_class("primary-result-context") for node in _children(bodies[1])): + evidence_shape.append(("section", "primary-result-context")) + evidence_shape.append(("nav", "evidence-nav")) evidence_body = _shape(bodies[1], evidence_shape) if _plain(evidence_body[0]) != "What the evidence shows": raise ValueError("Unknown evidence heading.") diff --git a/README.md b/README.md index ccaf029..476ac06 100644 --- a/README.md +++ b/README.md @@ -157,9 +157,11 @@ one side. Each run creates a local HTML report with five tabs: - **Summary** tells a numbered story: the intended change, what the evidence shows, and what it means. Illustrated Before/After cards retain exact counts - and distinguish plans, answers, and recorded actions. Mixed results and - evidence gaps remain visible. **View this comparison** opens the supporting - comparison, or links to trial records when extraction is unavailable. + and distinguish plans, answers, and recorded actions. When the cards focus on + an explanation or another secondary comparison, the primary result and its + full Before/After distribution remain visible beside them. Mixed results and + evidence gaps remain visible. One **View behavior comparisons** button opens + Behavior diff for both the featured comparison and the primary result. **Full scenario and expected behavior** explains the simulated situation, instruction versions, trial setup, and supplied expectation or its absence. **View full scenario prompt** is nested inside that disclosure. @@ -214,8 +216,11 @@ come from recorded commands or self-reported actions; **Answer detail** comparis come from the final answer. Wording differences alone do not establish an action change. In Summary and Behavior diff, counts such as **3 of 3 trials** refer to trials, -not repeated actions within one trial. A model extracts these counts from the -evidence. Separate row counts do not show a complete sequence within one trial. +not repeated actions within one trial. For new extractions, the model assigns +named trials to behaviors; code derives counts from complete, unique membership +on each side. Those assignments remain in `decisions.json` for auditing. +This validates bookkeeping, not whether a trial was classified correctly. +Separate row counts do not show a complete sequence within one trial. **Changed** compares extracted behavior proportions. **Unchanged** is shown for complete, unanimous same-behavior evidence; matching proportions without that evidence are labeled **Same proportions**. **Unavailable** means extracted @@ -233,7 +238,7 @@ The command-category table groups each trial into one category combination. Category order is not execution order. Matching categories can contain different commands or files, so matching patterns do not establish unchanged behavior. -Follow **View this comparison** to Behavior diff, or inspect the trial records +Follow **View behavior comparisons** to Behavior diff, or inspect the trial records for original commands and answers. Markdown retains the five sections, story, counts, complete instruction diff, and grouped trial evidence without illustrations. @@ -243,6 +248,10 @@ interprets their relationship to the supplied edit. **Related edit** links are labeled as model interpretation, not causal proof or knowledge of author intent. The same call supplies optional plain-language Summary text: a takeaway, short scenario, Before/After descriptions, and supported implications or cautions. +The writing instructions require concrete actor/action/object contrasts, parallel +Before/After descriptions, and a clear distinction between changed decisions or +actions and changed explanations, citations, or presentation. The decisive +contrast belongs in the headline and cards, not only in an additional finding. It also interprets the edit's likely aim, citing instruction-diff hunks rather than inferring intent from trial outcomes. **Edit goal** shows one sentence with an **Inferred** badge and an edit link; the info popup explains the source diff --git a/docs/architecture.md b/docs/architecture.md index c4ac890..a4b3111 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -35,7 +35,7 @@ scripts handle execution, evidence, and reporting. v +---------------------------------------------------------+ | Evidence analysis | -| decisions.py: comparisons and trial summaries | +| decisions.py: trial assignments -> counts and summaries | | reporting/: deterministic comparison and report data | +----------------------------+----------------------------+ v @@ -189,13 +189,18 @@ The modules below live in the skill's [`scripts/`](../plugin/skills/behavior-dif - **Model-based interpretation:** `decisions.py` reads the task, trial actions, final answers, and numbered instruction-diff hunks. A separate model extracts - choices, counts, a primary result, supported implications, links to edits, - plain-language Summary text, the edit's likely aim, and short per-group trial - summaries in the same call. Each trial summary names its exact Before/After - records and separates answers and plans from recorded actions. - The inferred aim cites instruction hunks independently of trial outcomes. - The script validates references and exact choice coverage before writing - `decisions.json`. Invalid interpretation does not discard valid observations. + choices with named trial assignments, a primary result, supported implications, + links to edits, plain-language Summary text, the edit's likely aim, and short + per-group trial summaries in the same call. New raw branches supply `trials`, + not aggregate counts. The script requires each completed trial exactly once + per side of every row, derives `n`, and retains both in `decisions.json`. + Foreign, duplicate, incomplete, or missing assignments invalidate the row. + Count-only raw extraction is not accepted; saved normalized report evidence + remains renderable. Membership validation cannot prove classification truth. + Each trial summary names its exact Before/After records and separates answers + and plans from recorded actions. The inferred aim cites instruction hunks + independently of trial outcomes. The script validates references and exact + choice coverage; invalid interpretation does not discard valid observations. - **Deterministic assembly:** `reporting/load.py` reads saved evidence and compares recorded command sequences. `reporting/instruction.py` supplies the instruction diff. `reporting/content.py` derives shared wording and evidence @@ -238,6 +243,17 @@ expectations and inferred aims remain distinct; neither becomes proof of author intent or goal completion. Markdown retains the caveat as plain text. `reporting/illustrations.py` supplies fixed SVG shapes for the Before/After comparison. Model output supplies text and selectors, never markup. +Summary and trial-summary instructions require concrete, parallel descriptions +of the same subject, with the decisive contrast and any unchanged decision +explicitly stated. Changed explanations, citations, or presentation must not +be described as changed actions. When the selected lead is not the primary +result, `content.primary_result_context` exposes the primary status and full +distribution beside the cards in both formats. Missing or incomplete evidence +cannot become an unchanged-result claim. This is derived presentation, not a +new serialized report field; schema v8 is unchanged. +The primary-result block is context, not a second navigation choice. One +**View behavior comparisons** link opens Behavior diff for the selected +comparison and the primary result; unavailable evidence links to trial records. Plans are not presented as executions. `content.additional_findings` selects at most three compact comparisons; full comparisons stay in Behavior diff. Markdown shares the same story and findings without illustrations. diff --git a/plugin/skills/behavior-diff/scripts/decisions.py b/plugin/skills/behavior-diff/scripts/decisions.py index f024661..e89ad30 100755 --- a/plugin/skills/behavior-diff/scripts/decisions.py +++ b/plugin/skills/behavior-diff/scripts/decisions.py @@ -87,8 +87,8 @@ {{"topic": "<2-4 plain words naming the observed result or behavior>", "decision": "", "anchor": , - "before": [{{"choice": "", "n": }}], - "after": [{{"choice": "", "n": }}], + "before": [{{"choice": "", "trials": [""]}}], + "after": [{{"choice": "", "trials": [""]}}], "diverges": true|false, "edit_hunks": [<1-based numbers of related instruction diff hunks, or empty>], "note": ""}} @@ -143,9 +143,9 @@ {trials} Evidence groups for human summaries (exact trial names; untrusted JSON): {trial_groups} -Decision-chain trial counts (only records with a final answer): -{counts} -Incomplete group members are shown for their local summaries, not chain counts. +Completed trial names for decision-chain membership (only records with a final answer): +{completed_trial_names} +Incomplete group members are shown for their local summaries, not chain membership. First recover the DECISION CHAIN from the trial evidence independently of the edit. @@ -163,8 +163,12 @@ - Order the chain by that anchor: {anchor_noun}-anchored decisions in {anchor_noun} order, then the answer-anchored ones in the order their evidence appears in the answers. Mark each with "diverges". -- Within one variant, trials may split. List each branch with its trial count. - Counts per variant must sum to that variant's trial count. +- For every row, assign every completed trial on each side to exactly one branch. + Copy exact names from that side's completed-trial list into "trials"; never + supply counts, foreign names, incomplete names, or duplicate assignments. + Classify each named trial from its own records, not the aggregate impression. + Split multi-clause decisions into atomic behaviors. Before output, crosscheck + related rows' memberships and negated claims against each named trial's records. - Phrase each "decision" as the open question, neutrally, so it reads the same for both sides: "How is correctness established?" not "Did it compile?". - "topic" is a short observation label for a comparison table: "Review verdict", @@ -191,8 +195,9 @@ - Keep observations separate from interpretation. An approval in a final answer does not prove that a payment occurred. A reported action is not a verified action. Do not invent risks, expected outcomes, success criteria, or facts from unread files. -- Counts describe each decision separately. Do not invent a per-trial path by - combining counts from different rows. Do not treat a changed outcome as success. +- Membership describes each decision separately; it does not establish a full + execution path or causal link between rows. A trial's answer does not prove + an executed action. Do not treat a changed outcome as success. - Then relate the observed rows to the numbered instruction hunks below. Set "edit_hunks" only when the hunk's changed lines concern the behavior actually observed in that row. Use each matching hunk number once. @@ -216,6 +221,14 @@ exactly. Do not provide summary counts: the application uses validated row counts. Mixed primary results must not be described as unanimous even when the lead is another action. Same-result/different-process is a valid finding. +- Write concrete actor + verb + object sentences in plain language. Explain an + internal workflow name only when the reader needs it to understand the finding. + Use parallel before/after sentences about the same subject; say what changed + and what stayed the same. Distinguish changed choices, actions, or stated plans + from changed explanations, citations, or presentation alone. Do not infer + actions from answers. Put the decisive contrast in the headline and main side + details, not only why/caution; avoid abstract correlation or the author's stance. + Do not duplicate count claims in prose: the application derives counts. - Use "plans" only for plans stated in final answers, "answers" for other final answer choices, and "actions" only for a numbered action/command anchor. Plans are not executed actions. Final answers do not prove tool execution. Self-reported @@ -237,6 +250,12 @@ side, preferably at most 25 words each. Every text field must be at most 240 characters and 40 words. Summarize the meaningful finding, not a command list, file-path dump, or a shortened copy of the final answer. Use plain text only. +- Use concrete actor + verb + object sentences and parallel side sentences about + the same subject. State what changed and what stayed the same, including when + choices stayed the same but explanations, citations, or presentation changed. + Put the decisive contrast in the takeaway and side sentences, not just caveat. + Explain internal workflow names only when necessary; avoid abstract correlation, + author stance, and duplicate count claims. Do not infer actions from answers. - Distinguish recorded actions from final-answer claims and plans. A final answer saying it tested something is not recorded testing. Self-reported actions must be described as reported, not verified. Recorded commands do not prove success. @@ -328,8 +347,23 @@ def read_config(run): return config -def normalize(data, counts, hunk_count=0, groups=()): - """Validate observations and links; remap row references after sorting or dropping.""" +def normalize(data, completed_trial_names, hunk_count=0, groups=()): + """Validate exact trial partitions and links; remap sorted or dropped row citations.""" + expected = {} + for side in ("before", "after"): + names = completed_trial_names.get(side) + if ( + type(names) not in (list, tuple) + or not names + or any( + type(name) is not str or not name.startswith(side + "-") + for name in names + ) + or len(set(names)) != len(names) + ): + raise ValueError(f"invalid completed {side} trial names") + expected[side] = set(names) + counts = {side: len(names) for side, names in expected.items()} if type(data) is not dict or type(data.get("chain", [])) is not list: raise ValueError("expected a decision chain") chain, dropped = [], 0 @@ -354,23 +388,33 @@ def normalize(data, counts, hunk_count=0, groups=()): branches = row.get(variant) if type(branches) is not list or not branches: break - choices = {} + choices, assigned = {}, set() for branch in branches: if ( type(branch) is not dict or type(branch.get("choice")) is not str or not branch["choice"].strip() - or type(branch.get("n")) is not int - or branch["n"] <= 0 or branch["choice"].strip() in choices + or type(branch.get("trials")) is not list + or not branch["trials"] + or any(type(name) is not str for name in branch["trials"]) + ): + break + names = branch["trials"] + members = set(names) + if ( + len(members) != len(names) + or not members <= expected[variant] + or members & assigned ): break - choices[branch["choice"].strip()] = branch["n"] + assigned.update(members) + choices[branch["choice"].strip()] = names else: - if sum(choices.values()) == counts.get(variant, 0): + if assigned == expected[variant]: clean[variant] = [ - {"choice": choice, "n": count} - for choice, count in choices.items() + {"choice": choice, "trials": list(names), "n": len(names)} + for choice, names in choices.items() ] continue break @@ -587,14 +631,17 @@ def run_extractor(prompt, agent=None, model=None): def build_prompt(run): - """Return (prompt, counts, instruction_diff), with no prompt for incomplete trials. + """Return (prompt, completed_trial_names, instruction_diff). Every extraction mode uses the same evidence and hunk numbering. """ trials = trials_of(run) - counts = {v: len(trials.get(v, [])) for v in ("before", "after")} - if not counts["before"] or not counts["after"]: - return None, counts, "" + completed_trial_names = { + side: [trial["name"] for trial in trials.get(side, [])] + for side in ("before", "after") + } + if not completed_trial_names["before"] or not completed_trial_names["after"]: + return None, completed_trial_names, "" all_trials = trials_of(run, finished_only=False) task = (run / "task.md").read_text().strip() config = read_config(run) @@ -618,25 +665,25 @@ def build_prompt(run): ensure_ascii=False, indent=2, ), - counts=json.dumps(counts), + completed_trial_names=json.dumps(completed_trial_names), instruction_hunks=json.dumps(hunks, ensure_ascii=False, indent=2), schema=schema, anchor_noun=terms["anchor_noun"], evidence_clause=terms["evidence_clause"], decision_clause=terms["decision_clause"], ), - counts, + completed_trial_names, instruction_diff, ) -def write_decisions(run, raw, counts, extractor, instruction_diff): +def write_decisions(run, raw, completed_trial_names, extractor, instruction_diff): """extract_json → normalize → decisions.json. On unusable output writes decisions.raw.txt, leaves decisions.json unwritten, returns False.""" try: data = normalize( extract_json(raw), - counts, + completed_trial_names, len(parse_diff_hunks(instruction_diff)), summary_groups(trials_of(run, finished_only=False)), ) @@ -649,12 +696,12 @@ def write_decisions(run, raw, counts, extractor, instruction_diff): (run / "decisions.raw.txt").write_text(raw) return False data["extractor"] = extractor - data["counts"] = counts + data["counts"] = {side: len(names) for side, names in completed_trial_names.items()} data["instruction_diff"] = instruction_diff (run / "decisions.json").write_text(json.dumps(data, indent=2)) n_div = sum(r["diverges"] for r in data["chain"]) drop_note = ( - f", {data['dropped']} row(s) dropped for inconsistent counts" + f", {data['dropped']} row(s) dropped for inconsistent trial membership" if data["dropped"] else "" ) @@ -666,7 +713,7 @@ def write_decisions(run, raw, counts, extractor, instruction_diff): def main(run, agent=None, model=None): - prompt, counts, instruction_diff = build_prompt(run) + prompt, completed_trial_names, instruction_diff = build_prompt(run) if prompt is None: print(NEED_TRIALS) return @@ -674,7 +721,7 @@ def main(run, agent=None, model=None): if raw is None: print("decision diff: no extractor succeeded — skipped") return - write_decisions(run, raw, counts, extractor, instruction_diff) + write_decisions(run, raw, completed_trial_names, extractor, instruction_diff) def emit_prompt(run): @@ -690,7 +737,7 @@ def emit_prompt(run): def ingest(run, reply_file, label): """Use emitted diff provenance when present; direct ingest uses the current diff.""" - prompt, counts, instruction_diff = build_prompt(run) + prompt, completed_trial_names, instruction_diff = build_prompt(run) if prompt is None: sys.exit(NEED_TRIALS) context_path = run / "decisions.prompt.json" @@ -703,7 +750,7 @@ def ingest(run, reply_file, label): except (OSError, ValueError, KeyError, TypeError): instruction_diff = "" raw = Path(reply_file).read_text() - if not write_decisions(run, raw, counts, label, instruction_diff): + if not write_decisions(run, raw, completed_trial_names, label, instruction_diff): sys.exit(1) @@ -712,14 +759,20 @@ def progress(message): print(f"[decisions] {message}", flush=True) progress("Normalize decision rows and parse extractor JSON") - counts = {"before": 3, "after": 3} + counts = { + "before": ["before-1", "before-2", "before-3"], + "after": ["after-1", "after-2", "after-3"], + } good = { "chain": [ { "decision": "What verdict shape?", "anchor": "answer", - "before": [{"choice": "PASS", "n": 2}, {"choice": "FAIL", "n": 1}], - "after": [{"choice": "score /100", "n": 3}], + "before": [ + {"choice": "PASS", "trials": counts["before"][:2]}, + {"choice": "FAIL", "trials": counts["before"][2:]}, + ], + "after": [{"choice": "score /100", "trials": counts["after"]}], "diverges": True, "edit_hunks": [2], }, @@ -727,16 +780,16 @@ def progress(message): "decision": "How is correctness established?", "anchor": 2, "topic": "Correctness check", - "before": [{"choice": "ran the program", "n": 3}], - "after": [{"choice": "traced by hand", "n": 3}], + "before": [{"choice": "ran the program", "trials": counts["before"]}], + "after": [{"choice": "traced by hand", "trials": counts["after"]}], "diverges": True, "edit_hunks": [1, 2], }, { - "decision": "bad row, counts do not add up", + "decision": "bad row, membership is incomplete", "anchor": 1, - "before": [{"choice": "x", "n": 1}], - "after": [{"choice": "y", "n": 3}], + "before": [{"choice": "x", "trials": counts["before"][:1]}], + "after": [{"choice": "y", "trials": counts["after"]}], "diverges": False, }, ], @@ -751,6 +804,72 @@ def progress(message): {"text": "Unsupported claim from a dropped row.", "decisions": [1, 3]}, ], } + # Synthetic memberships establish counts; model-authored n cannot override them. + inflated = { + **good, + "chain": [ + { + **good["chain"][1], + "before": [{**good["chain"][1]["before"][0], "n": 999}], + } + ], + } + audited = normalize(inflated, counts) + assert audited["chain"][0]["before"] == [ + {"choice": "ran the program", "trials": counts["before"], "n": 3} + ] + assert audited["chain"][0]["after"][0]["n"] == 3 + for invalid_names in ( + {"before": 3, "after": 3}, + {**counts, "before": []}, + {**counts, "before": [*counts["before"], "before-1"]}, + {**counts, "before": counts["after"]}, + ): + try: + normalize(good, invalid_names) + except ValueError: + pass + else: + raise AssertionError("Invalid completed trial identities were accepted") + for side in ("before", "after"): + opposite = "after" if side == "before" else "before" + malformed_assignments = [ + None, + [], + [{"choice": "A", "trials": side + "-1"}], + [{"choice": "A", "trials": [True]}], + [{"choice": "A", "trials": []}], + [{"choice": "A", "trials": counts[side][:2]}], + [{"choice": "A", "trials": [*counts[side], side + "-foreign"]}], + [{"choice": "A", "trials": counts[opposite]}], + [{"choice": "A", "trials": [*counts[side], counts[side][0]]}], + [ + {"choice": "A", "trials": counts[side]}, + {"choice": "B", "trials": counts[side][:1]}, + ], + [ + {"choice": "A", "trials": counts[side][:2]}, + {"choice": "A", "trials": counts[side][2:]}, + ], + ] + for branches in malformed_assignments: + candidate = { + **good, + "chain": [{**good["chain"][0], side: branches}, good["chain"][1]], + "outcome": 1, + "implications": [ + {"text": "Lost evidence.", "decisions": [1, 2]}, + {"text": "Retained evidence.", "decisions": [2]}, + ], + } + rejected = normalize(candidate, counts) + assert rejected["dropped"] == 1, rejected + assert len(rejected["chain"]) == 1, rejected + assert rejected["fork"] == 1, rejected + assert rejected["outcome"] is None, rejected + assert rejected["implications"] == [ + {"text": "Retained evidence.", "decisions": [1]} + ], rejected out = normalize(good, counts, 2) assert len(out["chain"]) == 2, out # bad row dropped assert out["dropped"] == 1, out # ...and counted, not silent @@ -769,7 +888,7 @@ def progress(message): invalid = {**good, "chain": [{**good["chain"][1], "edit_hunks": references}]} normalized = normalize(invalid, counts, 2) assert normalized["chain"][0]["edit_hunks"] == [], normalized - assert normalized["chain"][0]["before"] == good["chain"][1]["before"] + assert normalized["chain"][0]["before"] == out["chain"][0]["before"] assert normalized["dropped"] == 0 # Edit citations never follow sorted/dropped decision indexes. @@ -839,13 +958,13 @@ def progress(message): normalized = normalize(invalid, counts) assert normalized["outcome"] is None assert normalized["implications"] == [] - # Neither negative counts nor the model's divergence flag can imply a valid change. + # Count-only raw extraction is not a compatibility path. invalid_counts = { "chain": [ { "decision": "What outcome?", "anchor": "answer", - "before": [{"choice": "A", "n": 4}, {"choice": "B", "n": -1}], + "before": [{"choice": "A", "n": 3}], "after": [{"choice": "A", "n": 3}], } ] @@ -856,15 +975,15 @@ def progress(message): { "decision": "What outcome?", "anchor": "answer", - "before": [{"choice": "A", "n": 2}], - "after": [{"choice": "A", "n": 3}], + "before": [{"choice": "A", "trials": counts["before"][:2]}], + "after": [{"choice": "A", "trials": counts["after"]}], "diverges": True, "edit_hunks": [1], } ], "fork": 1, } - normalized = normalize(proportional, {"before": 2, "after": 3}, 1) + normalized = normalize(proportional, {**counts, "before": counts["before"][:2]}, 1) assert not normalized["chain"][0]["diverges"] assert normalized["fork"] is None assert normalized["chain"][0]["edit_hunks"] == [1] @@ -1140,8 +1259,10 @@ def cli(*argv, env=None): "topic": "Verdict shape", "decision": "What verdict shape?", "anchor": "answer", - "before": [{"choice": "prose", "n": 1}], - "after": [{"choice": "flagged item", "n": 1}], + "before": [{"choice": "prose", "trials": ["before-1"]}], + "after": [ + {"choice": "flagged item", "trials": ["after-1"]} + ], "diverges": True, "edit_hunks": [1], } @@ -1163,6 +1284,12 @@ def cli(*argv, env=None): assert len(data["chain"]) == 1, data assert data["extractor"] == "subagent:sonnet", data assert data["counts"] == {"before": 1, "after": 1}, data + assert data["chain"][0]["before"] == [ + {"choice": "prose", "trials": ["before-1"], "n": 1} + ], data + assert data["chain"][0]["after"] == [ + {"choice": "flagged item", "trials": ["after-1"], "n": 1} + ], data assert data["chain"][0]["edit_hunks"] == [] assert data["intent"] is None @@ -1313,8 +1440,8 @@ def cli(*argv, env=None): { "decision": "Which result?", "anchor": "answer", - "before": [{"choice": "before", "n": 1}], - "after": [{"choice": "after", "n": 1}], + "before": [{"choice": "before", "trials": ["before-1"]}], + "after": [{"choice": "after", "trials": ["after-1"]}], "diverges": True, } ], diff --git a/plugin/skills/behavior-diff/scripts/reporting/content.py b/plugin/skills/behavior-diff/scripts/reporting/content.py index 5290af6..28d82c4 100644 --- a/plugin/skills/behavior-diff/scripts/reporting/content.py +++ b/plugin/skills/behavior-diff/scripts/reporting/content.py @@ -1,6 +1,8 @@ """Format-neutral wording for Behavior Diff reports.""" from collections import Counter +from dataclasses import dataclass +from typing import Optional, Tuple from reporting.instruction import parse_diff_hunks from reporting.schema import ContentData, ResultData @@ -215,6 +217,70 @@ def intent_context(intent): ) +@dataclass(frozen=True) +class PrimaryResultContext: + heading: str + status: str + decision: Optional[int] + sides: Tuple[Tuple[str, str], ...] + note: str + + +def primary_result_context(report): + """Keep a secondary lead from obscuring the canonical primary result.""" + decisions = report.decisions + primary = decisions.outcome + if primary is not None and primary == report.summary.decision: + return None + if primary is None: + return PrimaryResultContext( + "Primary result", + "unavailable", + None, + (), + "No primary result was identified; the selected comparison does not establish one.", + ) + row = decisions.rows[primary - 1] + status = "unavailable" + if complete_trial_evidence(report): + unanimous = unanimous_choices(row, decisions) + status = ( + "mixed" + if unanimous is None + else "unchanged" + if unanimous[0] == unanimous[1] + else "changed" + ) + sides = tuple( + ( + label, + "; ".join( + "{0} ({1})".format(choice.choice, trial_count(choice.count, total)) + for choice in choices + ) + or NO_EXTRACTED_CHOICE, + ) + for label, choices, total in ( + ("Before", row.before, decisions.before_count), + ("After", row.after, decisions.after_count), + ) + ) + note = ( + "Based on reported final answers; these outcomes do not prove execution." + if row.anchor == "answer" + else "Model-extracted outcomes do not establish successful execution." + ) + if status == "unavailable": + note += " Evidence is incomplete; these counts do not establish a result comparison." + return PrimaryResultContext( + "Reported primary result" if row.anchor == "answer" else "Primary result", + status, + primary, + sides, + note, + ) + + NO_ADDITIONAL_FINDINGS = ( "No additional supported findings were selected for this summary." ) diff --git a/plugin/skills/behavior-diff/scripts/reporting/render_html.py b/plugin/skills/behavior-diff/scripts/reporting/render_html.py index 79404b0..c879abd 100644 --- a/plugin/skills/behavior-diff/scripts/reporting/render_html.py +++ b/plugin/skills/behavior-diff/scripts/reporting/render_html.py @@ -513,6 +513,21 @@ def _summary_side(side, label: str) -> str: ) +def _primary_result_context(report) -> str: + context = content.primary_result_context(report) + if context is None: + return "" + sides = "".join( + f"

{html.escape(label)}: {html.escape(text)}

" + for label, text in context.sides + ) + return ( + '
' + f"

{html.escape(context.heading)} — {html.escape(context.status)}

" + f'{sides}

{html.escape(context.note)}

' + ) + + def _short_story_summary(report: ReportData) -> str: summary = report.summary intent = report.intent @@ -538,12 +553,10 @@ def _short_story_summary(report: ReportData) -> str: comparison_available = ( summary.decision is not None and summary.status != "unavailable" ) - destination = ( - f"#decision-{summary.decision}" if comparison_available else "#panel-trials" - ) + destination = "#panel-decision" if comparison_available else "#panel-trials" lead_link = ( f'' - f"{'View this comparison' if comparison_available else 'View available trial records'}" + f"{'View behavior comparisons' if comparison_available else 'View available trial records'}" ' ' ) notices = "".join(f"
  • {html.escape(note)}
  • " for note in summary.notices) @@ -573,6 +586,7 @@ def _short_story_summary(report: ReportData) -> str: f"{_summary_side(summary.before, 'Before')}" '' f"{_summary_side(summary.after, 'After')}" + f"{_primary_result_context(report)}" '" "" diff --git a/plugin/skills/behavior-diff/scripts/reporting/render_markdown.py b/plugin/skills/behavior-diff/scripts/reporting/render_markdown.py index 914b4ee..e0f12ae 100644 --- a/plugin/skills/behavior-diff/scripts/reporting/render_markdown.py +++ b/plugin/skills/behavior-diff/scripts/reporting/render_markdown.py @@ -285,13 +285,19 @@ def _summary_markdown(report): if choice.detail: markdown.append(_text(choice.detail) + "\n") markdown.append(_text(content.trial_count(choice.count, side.total)) + "\n") + context = content.primary_result_context(report) + if context is not None: + markdown.append(f"#### {_text(context.heading)} — {_text(context.status)}\n") + for label, text in context.sides: + markdown.append(f"**{label}:** {_text(text)}\n") + markdown.append(_text(context.note) + "\n") target = ( - f"decision-{summary.decision}" + "panel-decision" if summary.decision is not None and summary.status != "unavailable" else "panel-trials" ) label = ( - "View this comparison" + "View behavior comparisons" if target != "panel-trials" else "View available trial records" ) diff --git a/plugin/skills/behavior-diff/scripts/reporting/report.css b/plugin/skills/behavior-diff/scripts/reporting/report.css index ce33f94..a4d84f8 100644 --- a/plugin/skills/behavior-diff/scripts/reporting/report.css +++ b/plugin/skills/behavior-diff/scripts/reporting/report.css @@ -421,6 +421,12 @@ details { margin-bottom:.25rem; } .summary-count { display:block; font-size:12px; font-weight:700; color:#142b35; font-variant-numeric:tabular-nums; } .summary-arrow { align-self:center; text-align:center; font-size:30px; color:#758790; } +.primary-result-context { border:1px solid #dbe5e8; border-radius:10px; + background:#f5f8fa; padding:14px 18px; margin:16px 0; } +.primary-result-context h4 { margin:0 0 10px; font-size:16px; } +.primary-result-context p { margin:6px 0; font-size:14px; line-height:1.5; } +.primary-result-context a { display:inline-block; padding:.35rem 0; + font-size:13px; font-weight:700; } .summary-why, .summary-caution { display:grid; grid-template-columns:110px minmax(0, 1fr); align-items:baseline; gap:16px; margin:0 0 14px; } .summary-why strong, .summary-caution strong { font-size:13px; } diff --git a/tests/human-evaluation-quiz-test.py b/tests/human-evaluation-quiz-test.py index c86bba6..4f16432 100755 --- a/tests/human-evaluation-quiz-test.py +++ b/tests/human-evaluation-quiz-test.py @@ -354,6 +354,47 @@ def test_actual_synthetic_report_blinding_preserves_wording(self): quiz._hash(original.encode()), ) + def test_blinding_retains_primary_result_beside_answer_detail_contrast(self): + for scenario in ("answer-details", "non-outcome-narrative", "missing-primary"): + with self.subTest(scenario=scenario): + source = (self.gallery / scenario / "report.html").read_text() + original = quiz._ReportParser().finish(source) + context = quiz._single( + [ + node + for node in quiz._walk(original) + if node.has_class("primary-result-context") + ], + "primary result", + ) + blinded = quiz._ReportParser().finish(quiz._blinded_report(source)) + retained = quiz._single( + [ + node + for node in quiz._walk(blinded) + if node.has_class("primary-result-context") + ], + "blinded primary result", + ) + self.assertEqual( + [ + quiz._plain(node) + for node in context.children + if isinstance(node, quiz._Node) and node.tag in ("h4", "p") + ], + [ + quiz._plain(node) + for node in retained.children + if isinstance(node, quiz._Node) and node.tag in ("h4", "p") + ], + ) + self.assertFalse( + any( + node.tag == "a" or "href" in node.attrs + for node in quiz._walk(retained) + ) + ) + def test_all_saved_report_shapes_and_unavailable_extraction(self): for report in self.gallery.glob("*/report.html"): with self.subTest(scenario=report.parent.name): diff --git a/tests/report-schema-test.py b/tests/report-schema-test.py index 7876fe1..18b646f 100755 --- a/tests/report-schema-test.py +++ b/tests/report-schema-test.py @@ -632,7 +632,7 @@ def handle_endtag(self, tag): ) if is_html: assert evidence.primary_targets == [ - f"#decision-{report.summary.decision}" + "#panel-decision" if report.summary.decision is not None and report.summary.status != "unavailable" else "#panel-trials" @@ -1170,6 +1170,105 @@ def summarize(variants=changed.variants, decisions=changed.decisions): assert_unchanged_scripts(render(changed), rendered) assert "<script>" in rendered assert "](javascript:" not in render_markdown(unsafe) + assert_primary_result_context(reports) + + +def assert_primary_result_context(reports): + """A secondary lead cannot conceal canonical primary outcomes or uncertainty.""" + from reporting.render_html import render_artifact + from reporting.render_markdown import render_markdown + from reporting.summary import build_summary + + base = reports["non-outcome-narrative"] + primary = base.decisions.outcome + original = base.decisions.rows[primary - 1] + + def with_primary(row=original, **changes): + rows = tuple( + row if index == primary else value + for index, value in enumerate(base.decisions.rows, 1) + ) + decisions = replace(base.decisions, rows=rows, **changes) + return replace( + base, + decisions=decisions, + summary=build_summary(base.metadata, base.variants, decisions), + ) + + same = with_primary(replace(original, after=original.before)) + assert same.summary.decision != primary + assert same.decisions.rows[same.summary.decision - 1].anchor == "answer" + mixed = with_primary( + replace( + original, + before=( + replace(original.before[0], count=2), + replace(original.after[0], count=1), + ), + ) + ) + no_primary = with_primary(outcome=None) + cases = ( + (same, "unchanged"), + (base, "changed"), + (mixed, "mixed"), + (with_primary(dropped=1), "unavailable"), + (no_primary, "unavailable"), + ) + for report, expected_status in cases: + # Incomplete evidence may select the primary fallback; force the existing + # secondary lead here to exercise the context's independent safety gate. + if report.summary.decision == report.decisions.outcome: + report = replace( + report, + summary=replace(report.summary, decision=base.summary.decision), + ) + context = content.primary_result_context(report) + assert context is not None and context.status == expected_status + assert context.decision == report.decisions.outcome + rendered = render_artifact(report, "") + block = rendered.split('class="primary-result-context"', 1)[1].split( + "", 1 + )[0] + markdown = render_markdown(report) + md_block = markdown.split(f"#### {context.heading}", 1)[1].split("### 3.", 1)[0] + assert expected_status in block and expected_status in md_block + assert "