From fa4f2ee2659e1fec2f7e24fb7894fc3718e401fa Mon Sep 17 00:00:00 2001 From: Kent Huang Date: Mon, 5 Oct 2026 21:43:20 +0800 Subject: [PATCH] feat: emphasize operative rules in behavior summaries Signed-off-by: Kent Huang --- .../SKILL.md | 22 ++- .../references/workflow.md | 114 ++++++++++++++-- docs/architecture.md | 13 +- .../skills/behavior-diff/scripts/decisions.py | 31 ++++- tests/report-schema-test.py | 69 ++++++++++ tests/report_fixtures.py | 126 ++++++++++++++++++ 6 files changed, 346 insertions(+), 29 deletions(-) diff --git a/.agents/skills/run-behavior-diff-human-evaluation/SKILL.md b/.agents/skills/run-behavior-diff-human-evaluation/SKILL.md index 781724a..b57ae24 100644 --- a/.agents/skills/run-behavior-diff-human-evaluation/SKILL.md +++ b/.agents/skills/run-behavior-diff-human-evaluation/SKILL.md @@ -44,9 +44,12 @@ plugin payload. Never trigger it from hooks, CI, or a scheduled job. For every case, prepare a neutral synthetic decision-point fixture and `scenario.json`, then four patch-grounded options in `question.json`. Prefer a fresh scenario worker that cannot see the options/answer key. - The correct statement must concern behavior this scenario can expose; - do not bundle unrelated changes. Keep unchanged or weak results. -6. Freeze all five fixtures and questions **before** any live trial: + The keyed statement must distinguish the anticipated After behavior from + Before at that decision point, not merely describe behavior true of both. + Complete the workflow's prefreeze contrast audit in the keyed rationale; + source support alone does not establish an observable contrast. Keep + unchanged or weak results. +6. Freeze all five fixtures and audited questions **before** any live trial: ```bash python3 "$EVAL" freeze "$SESSION" @@ -76,10 +79,15 @@ python3 "$EVAL" results "$SESSION" ``` If no submission exists, say so; never invent a score. Report correct out of -five, confidence, and insufficient-evidence count. Compare misses with the -saved scenarios, patches, and trial answers—not just the answer key. Separate -readability, factual fidelity, scenario coverage, and quiz ambiguity. An -explanation-only change is not necessarily a changed action or outcome. +five, confidence, and insufficient-evidence count. Assess question validity for +**all five cases** against saved Before/After answers, source sides, scenarios, +and blinded summaries using [the workflow](references/workflow.md#question-validity-assessment). +Distinguish an observed delta from an unchanged or both-sides match. Retain the +original score and explicitly qualify non-discriminating questions, whether +answered correctly or incorrectly; do not reinterpret them as comprehension +successes or failures. Separate readability, factual fidelity, scenario coverage, +and quiz ambiguity. An explanation-only change is not necessarily a changed +action or outcome. Chance averages 1.25/5. Five cases may share a skill; this score is not an estimate of overall product accuracy. Record methodology and aggregate diff --git a/.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md b/.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md index 1746c43..962f3ba 100644 --- a/.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md +++ b/.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md @@ -67,10 +67,10 @@ pinned parent revision. Do not inspect unrelated private company material. ```json { - "stem": "Which statement describes the instruction change?", + "stem": "Which statement describes the Before/After contrast in this scenario?", "scope": "The decision this scenario can expose; identify the relevant patch section.", "options": [ - {"statement": "First candidate statement.", "correct": true, "rationale": "Patch support and why the scenario covers it."}, + {"statement": "First candidate contrast.", "correct": true, "rationale": "Before: source support. After: changed rule. Scenario: triggering facts. Contrast: anticipated observable difference and limits."}, {"statement": "Second candidate statement.", "correct": false, "rationale": "Why the patch contradicts or does not introduce it."}, {"statement": "Third candidate statement.", "correct": false, "rationale": "Why this is false for the selected change."}, {"statement": "Fourth candidate statement.", "correct": false, "rationale": "Why this is false for the selected change."} @@ -79,15 +79,47 @@ pinned parent revision. Do not inspect unrelated private company material. ``` This is a schema example, not ready-to-use question content. Author four - concrete, similarly specific statements with exactly one supported by the - patch. Avoid bundled claims outside the scenario, trivia, obvious nonsense, + concrete, similarly specific statements with exactly one patch-grounded + contrast that the scenario can expose. The keyed statement must distinguish + anticipated After behavior from Before; a statement true of both sides is + not a discriminating answer, even if it accurately describes After. + Avoid bundled claims outside the scenario, trivia, obvious nonsense, uniquely repeated keywords, or an answer longer than all the distractors. - A clarification-only change is valid; do not claim it changes the outcome. - Check each rationale against the actual patch, independently of the report. -5. **Freeze all cases.** `freeze` validates inputs, shuffles options, writes - the private answer key and public questions, and records input hashes. + A clarification-only change can target a stated reason, condition, or timing; + do not claim it changes the action or outcome. Check each rationale against + both complete source sides and the patch, independently of the report. +5. **Complete the prefreeze contrast audit.** Before authorizing `freeze`, + record these four checks in the keyed option's existing `rationale`: + + - **Before:** cite the relevant source section and say whether the keyed + behavior is already required, permitted, or illustrated there. Check + surrounding rules and relevant references, not only removed patch lines. + - **After:** cite the changed rule and identify the precise added or changed + condition, timing, next step, or explanation. Separate explicit wording + from an inference about how a trial might respond. + - **Scenario:** identify the local facts and decision point that activate + that rule. State what the read-only task can and cannot demonstrate. + - **Contrast:** state what observable Before/After difference would support + the keyed statement and what would instead make it an unchanged or + both-sides match. A difference must concern the whole statement, not merely + a keyword appearing in After. + + Have the maintainer review all four checks before freezing all five cases. + Revise an unsupported or both-sides statement, or its scenario, only during + preparation, without seeing live results. Do not replace a sampled commit + because its rule is redundant or its expected contrast is weak. If no + defensible contrast can be authored, report the design limitation before + live execution rather than inventing a difference. +6. **Freeze all cases.** `freeze` validates input structure, shuffles options, + writes the private answer key and public questions, and records input hashes. + The existing schema is unchanged: `scope` bounds the claim and the keyed + `rationale` holds the audit. The deterministic helper enforces structure and + immutability, **not semantic contrast**; a successful freeze does not prove + that the question discriminates. The audit is a required maintainer gate. After freezing, do not edit fixtures, questions, or the key. Start a new evaluation if the design must change; retain the abandoned session and why. + Once results exist, a weak or absent contrast is evidence, never grounds to + reject, resample, retry, or change the keyed answer. Scenario design is model-assisted maintainer work, not a claim that the installed Behavior Diff plugin independently drafted the scenario. When delegating, keep @@ -147,11 +179,67 @@ source commit links, and the full reports become available afterward. Keep the service running until the human finishes; restart `serve` to resume later. `results SESSION` reads the saved submission without a model call. Report the -score out of five, confidence, insufficient-evidence count, and comments. -Investigate misses against the original reports, scenarios, patches, and traces. -Do not infer comprehension from keyword matching or score alone. Inspect factual -consistency even in correctly answered cases. Preserve the first score; do not -rescore after changing a question or revealing the answers. +original score out of five, confidence, insufficient-evidence count, and comments. +Then complete the question-validity assessment below for all cases, not just +misses. Do not infer comprehension from keyword matching or score alone. Preserve +the first submission and score; do not rescore after changing a question, +excluding weak cases, or revealing the answers. These instructions apply when +analyzing older sessions too; never retrofit their questions or frozen rationales. + +## Question validity assessment + +After submission, use only saved evidence. For every case, compare the frozen +keyed statement and rationale with both complete source sides, the patch, +scenario, original Before/After trial answers, and the blinded summaries that +the human saw. Record the assessment in private analysis notes alongside the +session, without editing hashed inputs, reports, the answer key, or submission. +No new model run is needed or authorized by this assessment. + +Separate two judgments: + +1. **Observed contrast:** does the whole keyed statement distinguish After from + Before in the saved answers? + - **Observed delta:** the answers support the stated Before/After distinction. + Name the changed condition, timing, next step, or explanation and any trial + variability; do not turn a partial pattern into a universal claim. + - **Unchanged / both-sides match:** the keyed behavior is present on both + sides, or the scoped behavior is unchanged. Explicitly label the question + **non-discriminating in this run**, even if the key is source-supported or + the human selected it correctly. + - **Not observed / contradicted:** usable answers do not show the forecast + contrast or instead show a different direction. Preserve that result. + - **Inconclusive:** missing, blocked, or inconsistent evidence prevents a + supported distinction. Name the missing evidence; do not infer a delta. +2. **Summary exposure:** do the blinded summaries faithfully expose the supported + distinction? Separate omitted operative rules or timing from factual errors, + scenario undercoverage, and question ambiguity. A delta visible only in full + traces does not prove the human could infer it from the quiz excerpt. + +Cite private evidence locations and describe what each side actually says. +Assess correctly answered cases as well as misses. Retain the raw keyed score +out of five and qualify non-discriminating, unobserved, or inconclusive cases in +the analysis. Do not count an unchanged match as evidence of successful delta +comprehension, or its missed key as evidence of poor comprehension. Report +validity counts separately from the score; do not publish a revised denominator +or retroactively choose a different correct option. + +### Synthetic table-and-wait example + +Suppose Before already documents a blast-radius table, and After adds a rule to +disclose that table before applying a change. In saved trials, both responses +show the table, but only After explicitly says it will wait before proceeding. +The keyed statement “After shows a blast-radius table” is a both-sides match, +not an observed delta. Preserve its original score and mark it non-discriminating. +The supported observed contrast is the stated wait/next-step difference, not +the presence of the table. A future question can target that timing distinction +only if its source-and-scenario audit supports it, and must be frozen before +its own trials; do not rewrite this session's key using the example. + +Keep source intent separate from observed behavior: a disclosure-before-action +rule does **not** by itself require literal user approval. If After says it +will wait for approval, report that as observed response wording, not as an +explicit source requirement unless the source actually contains that requirement. +A stated wait is not evidence that any command executed or approval was obtained. A random guess averages 1.25/5. Five possibly correlated cases cannot establish product-wide accuracy. Separate summary readability/fidelity from scenario diff --git a/docs/architecture.md b/docs/architecture.md index a4b3111..63d7e78 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -245,8 +245,17 @@ author intent or goal completion. Markdown retains the caveat as plain text. comparison. Model output supplies text and selectors, never markup. Summary and trial-summary instructions require concrete, parallel descriptions of the same subject, with the decisive contrast and any unchanged decision -explicitly stated. Changed explanations, citations, or presentation must not -be described as changed actions. When the selected lead is not the primary +explicitly stated. Summary selection prefers the observed changed operative rule +(conditions, timing, scope, or prerequisites) over its downstream outcome, including +mixed comparisons. Material intervals, deadlines, units, and gates belong in the +main cards, not only Other findings. Each card detail must hold for every named trial +in its selected branch; counts or citations from other rows cannot supply membership. +An instruction's gate is distinct from an observed gate, which may already appear +in Before. These are extraction policies, not deterministic semantic guarantees; +the existing validator checks references, choice coverage, and evidence anchors. +No schema field or keyword-based semantic check is added; fallback ranking is unchanged. +Changed explanations, citations, or presentation must not be described as changed +actions. When the selected lead is not the primary result, `content.primary_result_context` exposes the primary status and full distribution beside the cards in both formats. Missing or incomplete evidence cannot become an unchanged-result claim. This is derived presentation, not a diff --git a/plugin/skills/behavior-diff/scripts/decisions.py b/plugin/skills/behavior-diff/scripts/decisions.py index e89ad30..6b41071 100755 --- a/plugin/skills/behavior-diff/scripts/decisions.py +++ b/plugin/skills/behavior-diff/scripts/decisions.py @@ -212,23 +212,40 @@ decision observations, or presumed author motivation. Do not claim the aim was met. This edit-only model interpretation is independent of "summary" and chain indexes. - In the SAME reply, optionally give a concise plain-language "summary" grounded - in a meaningful selected chain row, or null when unsupported. Prefer a changed - primary result, then a unanimous changed action, then a mixed primary result, - then a changed mixed action or changed edit-linked comparison. Do not elevate - answer wording over a supported process difference. For no observed difference, - limit the headline to this scenario, never claim the edit has no effect. + in a meaningful selected chain row, or null when unsupported. Prefer the observed + changed operative rule: the condition, timing, scope, or prerequisite that changes + what the agent does or plans. Select that row rather than only its downstream + result, even when the primary result also changes or either row has mixed choices. + The application keeps the primary result and its full distribution beside the + cards. An edit link alone does not establish an observed rule change. + If no operative-rule contrast is supported, prefer a changed primary result, + then a unanimous changed action, then a mixed primary result, then a changed mixed + action or changed edit-linked comparison. Do not elevate answer wording over a + supported process difference. For no observed difference, limit the headline to + this scenario, never claim the edit has no effect. - Include EVERY choice on both sides of that row, copying its canonical "choice" exactly. Do not provide summary counts: the application uses validated row counts. Mixed primary results must not be described as unanimous even when the lead is another action. Same-result/different-process is a valid finding. +- Ground each side's label and detail in the records of EVERY named trial assigned + to that selected branch. Do not borrow a condition, interval, deadline, or other + fact from another branch or row's memberships, even when their counts match. + If a detail is not shared by those trials, split the observed choices faithfully, + select the row that records the rule, or omit that detail; do not combine rows + into a counted execution path. Why/caution citations do not support card details. - Write concrete actor + verb + object sentences in plain language. Explain an internal workflow name only when the reader needs it to understand the finding. Use parallel before/after sentences about the same subject; say what changed and what stayed the same. Distinguish changed choices, actions, or stated plans from changed explanations, citations, or presentation alone. Do not infer actions from answers. Put the decisive contrast in the headline and main side - details, not only why/caution; avoid abstract correlation or the author's stance. - Do not duplicate count claims in prose: the application derives counts. + details, not only why/caution or Other findings. Include material intervals, + deadlines and their units, triggers, prerequisites, and exceptions when supported; + a vague "waits longer" or "uses a stricter gate" is not enough. Describe the observed + gate separately from what the instruction requires: a newly written gate may + already appear in Before, and a rule in the diff is not evidence that After used it. + Avoid abstract correlation or the author's stance. Do not duplicate count claims + in prose: the application derives counts. - Use "plans" only for plans stated in final answers, "answers" for other final answer choices, and "actions" only for a numbered action/command anchor. Plans are not executed actions. Final answers do not prove tool execution. Self-reported diff --git a/tests/report-schema-test.py b/tests/report-schema-test.py index 18b646f..d22fbfa 100755 --- a/tests/report-schema-test.py +++ b/tests/report-schema-test.py @@ -1271,6 +1271,73 @@ def with_primary(row=original, **changes): assert content.primary_result_context(report).status == "unavailable" +def assert_timing_rule_cards(report): + """Rule cards retain timing and their own counts beside a mixed primary result.""" + from reporting.render_html import render_artifact + from reporting.render_markdown import render_markdown + from reporting.summary import build_summary + + assert report.summary.decision == 1 and report.decisions.outcome == 2 + assert report.summary.status == "mixed" + assert report.summary.evidence_label == ( + "Plans stated in final answers; not executed actions." + ) + primary = report.decisions.rows[1] + changed_primary = replace( + primary, + after=( + DecisionChoiceData("Recommend proceeding", 1), + DecisionChoiceData("Recommend deferring", 2), + ), + ) + decisions = replace( + report.decisions, rows=(report.decisions.rows[0], changed_primary) + ) + changed = replace( + report, + decisions=decisions, + summary=build_summary(report.metadata, report.variants, decisions), + ) + assert changed.summary.before == report.summary.before + assert changed.summary.after == report.summary.after + rendered = render_artifact(report, "") + changed_html = render_artifact(changed, "") + markdown = render_markdown(report) + cards = {} + md_cards = {} + for side, timing in ( + ("before", ("every 5 minutes", "15-minute deadline")), + ("after", ("every 10 minutes", "30-minute deadline", "check once")), + ): + marker = f'class="summary-card summary-{side}"' + cards[side] = html.unescape( + rendered.split(marker, 1)[1].split("", 1)[0] + ) + assert cards[side] == html.unescape( + changed_html.split(marker, 1)[1].split("", 1)[0] + ), "Primary-result counts must not rewrite operative-rule cards." + md_cards[side] = ( + markdown.split(f"#### {side.capitalize()}\n", 1)[1] + .split("#### ", 1)[0] + .replace(r"\-", "-") + ) + for text in ("wait for signoff", *timing): + assert text in cards[side] and text in md_cards[side] + assert "Recommend" not in cards[side] and "Recommend" not in md_cards[side] + for text in ("2 of 3 trials", "1 of 3 trials"): + assert text in cards["after"] and text in md_cards["after"] + context = content.primary_result_context(report) + changed_context = content.primary_result_context(changed) + assert context.status == changed_context.status == "mixed" + assert context.sides != changed_context.sides + assert all( + "Recommend proceeding" in text and "Recommend deferring" in text + for _, text in context.sides + ) + assert "primary result varies across trials" in rendered + assert "primary result varies across trials" in markdown + + def assert_additional_findings(reports): flip = reports["intent-flip"] findings = content.additional_findings(flip) @@ -1419,6 +1486,7 @@ def assert_gallery_reports(): "intent-flip": ("varies", "changed"), "intent-unchanged": ("unchanged", "unavailable"), "planned-actions": ("changed", "unavailable"), + "timing-rule": ("varies", "unavailable"), } with tempfile.TemporaryDirectory() as directory: root = Path(directory) @@ -1468,6 +1536,7 @@ def assert_gallery_reports(): assert reports["planned-actions"].decisions.trial_summaries[0].caveat == ( "Neither proposed step is recorded as executed." ) + assert_timing_rule_cards(reports["timing-rule"]) assert content.flow_patterns(reports["same-result"].command_flow) == ( (("Read files",), 3, 3), diff --git a/tests/report_fixtures.py b/tests/report_fixtures.py index 34a56f1..01dc873 100644 --- a/tests/report_fixtures.py +++ b/tests/report_fixtures.py @@ -100,6 +100,11 @@ class Scenario: "A different next step is proposed", "The answers propose a repair instead of another review; neither is executed.", ), + Scenario( + "timing-rule", + "Retry timing leads alongside mixed recommendations", + "Planned intervals and deadlines change; signoff remains required in both answers.", + ), ) @@ -1242,6 +1247,125 @@ def _planned_actions(run, scenario): ) +def _timing_rule(run, scenario): + rules = { + "before": ( + "# Retry review\n" + "State a retry plan: after signoff, poll every 5 minutes for 15 minutes.\n" + ), + "after": ( + "# Retry review\n" + "State a retry plan. Do not poll until signoff; then poll every " + "10 minutes for 30 minutes.\n" + ), + } + plans = ( + "After signoff, poll every 5 minutes for 15 minutes", + "After signoff, poll every 10 minutes for 30 minutes", + "After signoff, check once at the 30-minute deadline", + ) + details = ( + "It would wait for signoff, then poll every 5 minutes until a 15-minute deadline.", + "It would wait for signoff, then poll every 10 minutes until a 30-minute deadline.", + "It would wait for signoff, then check once at the 30-minute deadline.", + ) + trials = {"before": [], "after": []} + for side in trials: + for number in range(1, 4): + plan = 0 if side == "before" else 1 if number != 3 else 2 + deferred = number == (3 if side == "before" else 2) + recommendation = ( + "Recommend deferring" if deferred else "Recommend proceeding" + ) + trials[side].append( + _Trial( + actions=(("cat AGENTS.md", rules[side]),), + final=( + details[plan] + + " " + + recommendation + + " with this retry plan. No polling or signoff is recorded." + ), + choices={ + "Retry plan": plans[plan], + "Recommendation": recommendation, + }, + summary=details[plan], + ) + ) + _write_sources( + run, + scenario, + "Synthetic task: State your retry plan and recommend proceeding or deferring.", + rules, + {}, + trials, + ) + rows = [ + _row("Retry plan", "What retry plan is stated?", "answer", trials), + _row("Recommendation", "What is recommended?", "answer", trials), + ] + rows[0]["edit_hunks"] = [1] + _write_extraction( + run, + rows, + primary="Recommendation", + summary={ + "decision": 1, + "headline": "Retry plans extend the deadline; some use longer intervals.", + "scenario": "An agent proposes retry timing without executing the plan.", + "evidence_kind": "plans", + "before": { + "icon": "inspect", + "choices": [ + { + "choice": plans[0], + "label": "Plan five-minute polling", + "detail": details[0], + } + ], + }, + "after": { + "icon": "inspect", + "choices": [ + { + "choice": plans[index], + "label": label, + "detail": details[index], + } + for index, label in ( + (1, "Plan ten-minute polling"), + (2, "Plan one deadline check"), + ) + ], + }, + "why": None, + "caution": None, + }, + intent={ + "text": "Restate the signoff gate and extend retry intervals and deadline.", + "edit_hunks": [1], + }, + trial_summaries=_trial_summaries( + trials, + [ + ( + "Both plans wait for signoff; After extends the interval and deadline.", + "Neither retry plan is recorded as executed.", + ), + ( + "Both plans wait for signoff; After extends the interval and deadline.", + "Neither retry plan is recorded as executed.", + ), + ( + "Both plans wait for signoff; After proposes one later deadline check.", + "Neither retry plan is recorded as executed.", + ), + ], + ), + ) + + def build_reports(root: Path) -> None: """Build all catalog reports beneath an existing, empty directory. @@ -1268,6 +1392,8 @@ def build_reports(root: Path) -> None: _missing_primary(run, scenario) elif scenario.name == "planned-actions": _planned_actions(run, scenario) + elif scenario.name == "timing-rule": + _timing_rule(run, scenario) elif scenario.name in {"flow-changed", "flow-mixed"}: _availability_review(run, scenario) elif scenario.name in {"intent-flip", "intent-unchanged"}: