From ec001ffead18c3bdc208ab864cf683321c05f20b Mon Sep 17 00:00:00 2001 From: Kent Huang Date: Mon, 5 Oct 2026 17:20:08 +0800 Subject: [PATCH] feat: add manual human evaluation skill Add a private, fixed-source random evaluation workflow and blinded quiz for current local Behavior Diff summaries. Refs DRC-4795 Signed-off-by: Kent Huang --- .../SKILL.md | 87 + .../assets/index.html | 35 + .../assets/quiz.css | 31 + .../assets/quiz.js | 196 +++ .../references/workflow.md | 160 ++ .../scripts/evaluate.py | 1420 +++++++++++++++++ .../scripts/quiz.py | 980 ++++++++++++ .github/workflows/ci.yml | 6 + AGENTS.md | 5 + CODING_GUIDELINES.md | 2 + README.md | 31 + docs/architecture.md | 41 + tests/human-evaluation-quiz-test.py | 634 ++++++++ tests/human-evaluation-workflow-test.py | 711 +++++++++ 14 files changed, 4339 insertions(+) create mode 100644 .agents/skills/run-behavior-diff-human-evaluation/SKILL.md create mode 100644 .agents/skills/run-behavior-diff-human-evaluation/assets/index.html create mode 100644 .agents/skills/run-behavior-diff-human-evaluation/assets/quiz.css create mode 100644 .agents/skills/run-behavior-diff-human-evaluation/assets/quiz.js create mode 100644 .agents/skills/run-behavior-diff-human-evaluation/references/workflow.md create mode 100755 .agents/skills/run-behavior-diff-human-evaluation/scripts/evaluate.py create mode 100644 .agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py create mode 100755 tests/human-evaluation-quiz-test.py create mode 100755 tests/human-evaluation-workflow-test.py diff --git a/.agents/skills/run-behavior-diff-human-evaluation/SKILL.md b/.agents/skills/run-behavior-diff-human-evaluation/SKILL.md new file mode 100644 index 0000000..781724a --- /dev/null +++ b/.agents/skills/run-behavior-diff-human-evaluation/SKILL.md @@ -0,0 +1,87 @@ +--- +name: run-behavior-diff-human-evaluation +description: Run a private, five-question human evaluation of the current local Behavior Diff summaries using randomly sampled skill commits from DataRecce/recce-team. Use when a maintainer asks for a human evaluation, a blind report quiz, or to evaluate current summary ability; also use to resume or analyze an existing evaluation. +--- + +# Human evaluation of Behavior Diff + +Manually run five real comparisons, then ask a human to identify each change +from four shuffled statements. This is a repository-maintainer skill, not +plugin payload. Never trigger it from hooks, CI, or a scheduled job. + +## Start or resume + +1. For **analyze/resume**, locate the requested private session under + `~/.behavior-diff/human-evaluations/`; use its saved evidence. Never create + fresh trials to answer a request about an existing result. +2. For **new evaluation**, require this repository's checkout, Python 3.9+, + Git, authenticated `gh` access to `DataRecce/recce-team`, Bash, `jq`, and + the Claude Code CLI. + A Claude or Codex maintainer session can orchestrate it; the trial stack + is explicitly Claude Code with read-only tools, not the orchestrator's + default host. Read [the workflow](references/workflow.md) before starting. +3. Explain the spend and privacy boundary. Require fresh approval for this + evaluation: five reports, three Before and three After trials each, + plus extraction; private skill text is sent to the configured Claude + provider. Default selectors are `opus` for trials and `sonnet` for + extraction. Previous sessions' approvals do not authorize this one. +4. Initialize a fresh sample using the bundled helper: + + ```bash + EVAL=.agents/skills/run-behavior-diff-human-evaluation/scripts/evaluate.py + SESSION=$(python3 "$EVAL" init) + ``` + + Always use the fixed upstream `DataRecce/recce-team`. The helper pins its + current `main`, generates a new random seed, and samples five eligible + commits without replacement. Never handpick commits or reuse a saved + sample. Independent random samples may overlap; disclose known familiarity. + Keep SHAs, patches, subjects, and correct options out of human-facing chat. + +## Prepare and run + +5. Follow [case preparation](references/workflow.md#case-preparation). + For every case, prepare a neutral synthetic decision-point fixture and + `scenario.json`, then four patch-grounded options in `question.json`. + Prefer a fresh scenario worker that cannot see the options/answer key. + The correct statement must concern behavior this scenario can expose; + do not bundle unrelated changes. Keep unchanged or weak results. +6. Freeze all five fixtures and questions **before** any live trial: + + ```bash + python3 "$EVAL" freeze "$SESSION" + python3 "$EVAL" run "$SESSION" --approve-live + python3 "$EVAL" build "$SESSION" + python3 "$EVAL" serve "$SESSION" --port 0 + ``` + + Pass `--approve-live` only after step 3's approval. Use a managed long-lived + service for `serve`; give the human the printed loopback URL. The helper + uses this checkout's code, including uncommitted changes. Never switch to + the installed plugin, silently retry a case, or change code mid-evaluation. +7. Check actual trial completions, intended revision reads, and generated + report artifacts. A blocked or unchanged result is evidence, not grounds + for replacement. Verify the quiz in a browser without submitting answers. + Do not use the real session for a scoring smoke test. +8. Deliver the URL and instructions: read the summaries, choose one option + per case, rate confidence, flag insufficient evidence, and submit once. + Do not show ground truth until submission. Original reports and commit + links unlock afterward. Artifacts stay outside the checkout and are + never attached to a public issue or committed. + +## Analyze the human's answers + +```bash +python3 "$EVAL" results "$SESSION" +``` + +If no submission exists, say so; never invent a score. Report correct out of +five, confidence, and insufficient-evidence count. Compare misses with the +saved scenarios, patches, and trial answers—not just the answer key. Separate +readability, factual fidelity, scenario coverage, and quiz ambiguity. An +explanation-only change is not necessarily a changed action or outcome. + +Chance averages 1.25/5. Five cases may share a skill; this score is not an +estimate of overall product accuracy. Record methodology and aggregate +findings in the evaluation's tracking issue without private excerpts. Do not +rewrite summaries, edit answers, rerun models, or implement fixes unless asked. diff --git a/.agents/skills/run-behavior-diff-human-evaluation/assets/index.html b/.agents/skills/run-behavior-diff-human-evaluation/assets/index.html new file mode 100644 index 0000000..c3a34f9 --- /dev/null +++ b/.agents/skills/run-behavior-diff-human-evaluation/assets/index.html @@ -0,0 +1,35 @@ + + + + + + Behavior Diff — human evaluation + + + + +
+

Behavior Diff · Human evaluation

+

Read the summary. Choose the supported statement.

+

For each case, read the generated Before/After summary and select one of four statements. + Record your confidence. If the summary does not provide enough evidence to choose, flag that too; + still select your best answer. An optional note can explain ambiguity.

+

Your first complete submission is final. Answers, rationales, source commits and full reports + become available afterward. Draft answers are saved only in this browser for this session.

+
+
+

Loading this evaluation…

+ + +
+
These questions measure comprehension of the displayed reports in this session, not overall model accuracy.
+ + diff --git a/.agents/skills/run-behavior-diff-human-evaluation/assets/quiz.css b/.agents/skills/run-behavior-diff-human-evaluation/assets/quiz.css new file mode 100644 index 0000000..56e1366 --- /dev/null +++ b/.agents/skills/run-behavior-diff-human-evaluation/assets/quiz.css @@ -0,0 +1,31 @@ +:root { color-scheme: light; --ink: #1c2733; --muted: #526171; --border: #d9e1e7; --accent: #0b6e75; } +* { box-sizing: border-box; } +body { margin: 0 auto; max-width: 1140px; padding: 2rem 1.25rem; color: var(--ink); background: #f6f8f9; font: 16px/1.55 -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif; } +h1 { font-size: clamp(1.6rem, 5vw, 2.3rem); line-height: 1.25; margin-top: .25rem; } +h2, h3 { line-height: 1.35; } +.eyebrow { color: var(--accent); text-transform: uppercase; letter-spacing: .08em; font-size: .8rem; font-weight: 700; } +.notice { padding: 1rem; background: #e3f0f1; border-left: 3px solid var(--accent); } +.case { background: #fff; border: 1px solid var(--border); border-radius: .65rem; padding: 1.5rem; margin: 1.5rem 0; overflow-wrap: anywhere; } +.case > h2, .case > h3 { margin-top: 0; } +iframe { display: block; width: 100%; height: 600px; border: 1px solid var(--border); border-radius: .4rem; margin: 1rem 0 1.5rem; background: #f6f8f9; } +fieldset { border: 0; margin: 0 0 1rem; padding: 0; min-width: 0; } +legend { font-weight: 650; margin-bottom: .8rem; padding: 0; font-size: 1.1rem; } +.option { display: flex; align-items: flex-start; gap: .75rem; border: 1px solid var(--border); border-radius: .4rem; padding: .8rem; margin: .5rem 0; cursor: pointer; } +.option:has(input:checked) { border-color: var(--accent); background: #e3f0f1; } +input[type="radio"], input[type="checkbox"] { flex: none; margin: .35rem 0 0; width: 1.1rem; height: 1.1rem; accent-color: var(--accent); } +select, textarea, button { font: inherit; } +select, textarea { border: 1px solid #9baab8; border-radius: .3rem; padding: .5rem; color: var(--ink); background: #fff; max-width: 100%; } +textarea { display: block; width: 100%; resize: vertical; margin-top: .4rem; } +.sufficiency { display: flex; gap: .7rem; margin: 1rem 0; align-items: flex-start; } +.submit-row { display: flex; align-items: center; flex-wrap: wrap; gap: 1rem; padding: 1rem 0; } +button { padding: .75rem 1.2rem; border: 0; border-radius: .4rem; color: #fff; background: var(--accent); font-weight: 650; cursor: pointer; } +button:disabled { opacity: .6; cursor: default; } +a { color: var(--accent); text-underline-offset: .18em; } +a:focus-visible, button:focus-visible, input:focus-visible, select:focus-visible, textarea:focus-visible, summary:focus-visible { outline: 3px solid var(--accent); outline-offset: 3px; } +.links { display: flex; flex-wrap: wrap; gap: .7rem 1.2rem; } +summary { cursor: pointer; font-weight: 650; padding: .6rem 0; } +.score { font-size: 1.5rem; font-weight: 700; } +#status, #progress, footer { color: var(--muted); } +footer { margin-top: 2rem; border-top: 1px solid var(--border); padding-top: 1rem; } +[hidden] { display: none !important; } +@media (max-width: 600px) { body { padding: 1rem .6rem; } .case { padding: .8rem; } iframe { margin: .7rem 0 1rem; } .option { padding: .7rem; } } diff --git a/.agents/skills/run-behavior-diff-human-evaluation/assets/quiz.js b/.agents/skills/run-behavior-diff-human-evaluation/assets/quiz.js new file mode 100644 index 0000000..7885953 --- /dev/null +++ b/.agents/skills/run-behavior-diff-human-evaluation/assets/quiz.js @@ -0,0 +1,196 @@ +"use strict"; + +const byId = (id) => document.getElementById(id); +let sessionId; +let questions; +let storageKey; +let submitted = false; +const frames = new Map(); + +function element(tag, text, className) { + const node = document.createElement(tag); + if (text !== undefined) node.textContent = text; + if (className) node.className = className; + return node; +} + +async function request(path, options) { + const response = await fetch(path, { credentials: "same-origin", ...options }); + const data = await response.json(); + if (!response.ok) throw new Error(data.error || "The local server rejected this request."); + return data; +} + +function fitFrame(frame) { + try { + const doc = frame.contentDocument; + if (!doc || !doc.body) return; + // Measure content without changing the viewport inside ResizeObserver. + const style = frame.contentWindow.getComputedStyle(doc.body); + const margins = (parseFloat(style.marginTop) || 0) + (parseFloat(style.marginBottom) || 0); + const height = Math.ceil(doc.body.getBoundingClientRect().height + margins) + 24; + if (Math.abs((parseFloat(frame.style.height) || 0) - height) > 2) frame.style.height = `${height}px`; + } catch (_) { + // Full reports remain accessible through their separate post-submit link. + } +} + +function reportFrame(path, title, full = false) { + const frame = element("iframe"); + frame.title = title; + frame.src = path; + frame.setAttribute("sandbox", full ? "allow-scripts allow-same-origin" : "allow-same-origin"); + frame.addEventListener("load", () => { + fitFrame(frame); + const doc = frame.contentDocument; + if (!doc) return; + doc.addEventListener("toggle", () => requestAnimationFrame(() => fitFrame(frame)), true); + doc.addEventListener("click", () => setTimeout(() => fitFrame(frame), 40)); + if (window.ResizeObserver) { + const observer = new ResizeObserver(() => fitFrame(frame)); + observer.observe(doc.body); + frames.set(frame, observer); + } + }); + return frame; +} + +function answerFor(question) { + const id = question.id; + const selected = document.querySelector(`input[name="choice-${id}"]:checked`); + const confidence = byId(`confidence-${id}`).value; + return { id, letter: selected ? selected.value : "", confidence, + insufficient: byId(`insufficient-${id}`).checked, note: byId(`note-${id}`).value }; +} + +function saveDraft() { + if (submitted) return; + const answers = questions.map(answerFor); + try { localStorage.setItem(storageKey, JSON.stringify({ session_id: sessionId, answers })); } catch (_) { /* Storage is optional. */ } + const complete = answers.filter((answer) => answer.letter && answer.confidence).length; + byId("progress").textContent = `${complete} of ${questions.length} complete`; +} + +function restoreDraft() { + try { + const draft = JSON.parse(localStorage.getItem(storageKey)); + if (!draft || draft.session_id !== sessionId || !Array.isArray(draft.answers)) return; + for (const answer of draft.answers) { + if (!questions.some((question) => question.id === answer.id)) continue; + const choice = document.querySelector(`input[name="choice-${answer.id}"][value="${["A", "B", "C", "D"].includes(answer.letter) ? answer.letter : ""}"]`); + if (choice) choice.checked = true; + if (["low", "medium", "high"].includes(answer.confidence)) byId(`confidence-${answer.id}`).value = answer.confidence; + byId(`insufficient-${answer.id}`).checked = answer.insufficient === true; + if (typeof answer.note === "string") byId(`note-${answer.id}`).value = answer.note.slice(0, 2000); + } + } catch (_) { /* Ignore malformed or unavailable browser drafts. */ } +} + +function renderQuestions() { + for (const question of questions) { + const section = element("section", undefined, "case"); + section.append(element("h2", `Case ${question.id}`)); + section.append(reportFrame(`/summary-${question.id}.html`, `Blinded summary for case ${question.id}`)); + const fieldset = element("fieldset"); + fieldset.append(element("legend", question.stem)); + for (const option of question.options) { + const label = element("label", undefined, "option"); + const input = element("input"); + input.type = "radio"; input.name = `choice-${question.id}`; input.value = option.letter; input.required = true; + label.append(input, element("span", `${option.letter}. ${option.statement}`)); + fieldset.append(label); + } + section.append(fieldset); + const confidenceLabel = element("label", "Confidence "); + const confidence = element("select"); + confidence.id = `confidence-${question.id}`; confidence.required = true; + for (const [value, label] of [["", "Choose confidence"], ["low", "Low"], ["medium", "Medium"], ["high", "High"]]) { + const option = element("option", label); option.value = value; confidence.append(option); + } + confidenceLabel.append(confidence); + const insufficientLabel = element("label", undefined, "sufficiency"); + const insufficient = element("input"); + insufficient.type = "checkbox"; insufficient.id = `insufficient-${question.id}`; + insufficientLabel.append(insufficient, element("span", "The summary provides insufficient evidence to choose confidently.")); + const noteLabel = element("label", "Optional note"); + const note = element("textarea"); + note.id = `note-${question.id}`; note.maxLength = 2000; note.rows = 3; + noteLabel.append(note); + section.append(confidenceLabel, insufficientLabel, noteLabel); + byId("cases").append(section); + } + restoreDraft(); + byId("quiz").addEventListener("input", saveDraft); + byId("quiz").addEventListener("change", saveDraft); + saveDraft(); + byId("quiz").hidden = false; +} + +function renderResults(data) { + submitted = true; + byId("quiz").hidden = true; + byId("results").hidden = false; + byId("status").textContent = "First complete submission saved. You can now inspect the answer key and full reports."; + byId("score").replaceChildren(element("p", `${data.score.correct} of ${data.score.total} answers correct.`, "score"), element("p", data.interpretation)); + byId("reviews").replaceChildren(); + for (const answer of data.answers) { + const review = element("article", undefined, "case"); + review.append(element("h3", `Case ${answer.id}: ${answer.correct ? "correct" : "incorrect"}`)); + review.append(element("p", `Your answer: ${answer.letter}. Correct answer: ${answer.correct_letter}. Confidence: ${answer.confidence}. Insufficient evidence flagged: ${answer.insufficient ? "yes" : "no"}.`)); + review.append(element("p", `Question scope: ${answer.scope}`)); + const reasons = element("ul"); + for (const option of answer.options) { + const reason = element("li"); + reason.append(element("strong", `${option.letter}. ${option.statement}${option.correct ? " (correct)" : ""}`), element("p", option.rationale)); + reasons.append(reason); + } + review.append(reasons); + if (answer.note) review.append(element("p", `Your note: ${answer.note}`)); + const links = element("p", undefined, "links"); + for (const [url, label] of [[answer.source_url, "Source commit"], [answer.before_source_url, "Before commit"], [answer.review_url, "Open full report"]]) { + const link = element("a", label); link.href = url; link.target = "_blank"; link.rel = "noopener noreferrer"; links.append(link); + } + review.append(links); + const details = element("details"); + details.append(element("summary", "Full original Behavior Diff report")); + details.addEventListener("toggle", () => { + if (details.open && !details.querySelector("iframe")) details.append(reportFrame(answer.review_url, `Full report for case ${answer.id}`, true)); + else if (details.open) fitFrame(details.querySelector("iframe")); + }); + review.append(details); + byId("reviews").append(review); + } +} + +byId("quiz").addEventListener("submit", async (event) => { + event.preventDefault(); + if (submitted || !byId("quiz").reportValidity()) return; + byId("submit").disabled = true; + byId("status").textContent = "Saving your first complete submission…"; + try { + renderResults(await request("/submit", { method: "POST", headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ session_id: sessionId, answers: questions.map(answerFor) }) })); + } catch (error) { + byId("status").textContent = error.message; + byId("submit").disabled = false; + } +}); + +window.addEventListener("resize", () => { for (const frame of document.querySelectorAll("iframe")) fitFrame(frame); }); + +(async () => { + try { + const session = await request("/session-public.json"); + sessionId = session.session_id; + storageKey = `behavior-diff-human-evaluation:${sessionId}`; + // Server persistence, not browser storage, is authoritative about submission. + const response = await fetch("/results", { credentials: "same-origin" }); + if (response.ok) { renderResults(await response.json()); return; } + if (response.status !== 403) throw new Error("Unable to read this session's submission state."); + questions = await request("/questions.json"); + renderQuestions(); + byId("status").textContent = "Read all five summaries before submitting. Your draft is private to this session."; + } catch (error) { + byId("status").textContent = error.message; + } +})(); diff --git a/.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md b/.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md new file mode 100644 index 0000000..1746c43 --- /dev/null +++ b/.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md @@ -0,0 +1,160 @@ +# Human evaluation protocol + +Use this protocol only for an explicitly requested manual evaluation. It tests +how well a person can identify a skill change from generated behavior summaries. +It does not grade the upstream skills or estimate overall model accuracy. + +## Session and sampling + +The helper always clones `https://github.com/DataRecce/recce-team.git`, pins +`main`, and inspects its reachable nonmerge history. It never runs upstream +scripts, hooks, installers, or GitHub workflows. Each eligible commit must: + +- Have one parent and modify exactly one existing `SKILL.md`. +- Have no companion change in that skill's directory. +- Contain a non-whitespace change in the instruction file. + +The sample unit is a commit, not a skill. Five are drawn without replacement +from a freshly seeded random ordering. A repeated skill is allowed and must +be disclosed; do not swap it out to make results look more varied. Changes +outside the selected skill directory are not replayed. The inference scope +is the isolated instruction-file change, not the entire upstream commit. + +`session.json` records the source tip, seed, candidates, exclusions, selected +pairs, and local implementation identity. Reports measure the checkout's +current code, including uncommitted changes; the helper captures HEAD plus a +content fingerprint. A frozen session cannot silently use changed inputs. + +The default private storage is `~/.behavior-diff/human-evaluations/` with a +unique session directory. Never put it in a working tree, publicly served +folder, or shared/network-synced artifact directory. Do not upload snapshots, +questions, answer keys, reports, or submissions to issue trackers. Track the +method, consent, and aggregate findings instead. + +## Case preparation + +Each `case-N/` initially contains `before.md`, `after.md`, and `patch.diff`. +Read the complete affected workflow and directly relevant references at the +pinned parent revision. Do not inspect unrelated private company material. + +1. **Check replayability before results exist.** Formatting-only changes, + inseparable companion changes, or workflows that cannot be represented + safely with local records may be rejected with a specific reason. Use: + + ```bash + python3 "$EVAL" replace "$SESSION" --case 1 --reason "Concrete pre-run reason" + ``` + + Replacement uses the next unused member of the original random ordering, + not a chosen SHA. It is allowed only before freezing and before authored + case artifacts. A redundant rule, an unchanged result, or poor prose is + not grounds for rejection. If five eligible cases cannot be prepared, + report that limitation rather than inventing cases or changing repositories. +2. **Prepare `fixture/`.** Create a synthetic Git repository whose committed + target is byte-for-byte `before.md`. Use the original relative skill path + so relative references still work. Add only the synthetic task records and + necessary pinned reference text. Use a local synthetic Git identity and + a signed-off fixture commit. Then replace only the target with `after.md`; + it must be the only uncommitted change. No symlinks, upstream hooks, + credentials, real issue contents, or source repository `.git` directory. +3. **Write `scenario.json`.** It has `file` (the relative target path) and + `task` (the exact shared prompt). Explicitly ask the agent to read the + target skill and local records. Start at the decision point, request its + decision or draft, and state that commands, external services, edits, and + posting are unavailable. Do not state the expected answer. Trial tools + are restricted to Read/Grep/Glob; skill auto-discovery is disabled. +4. **Write `question.json` before seeing live results.** The shape is: + + ```json + { + "stem": "Which statement describes the instruction change?", + "scope": "The decision this scenario can expose; identify the relevant patch section.", + "options": [ + {"statement": "First candidate statement.", "correct": true, "rationale": "Patch support and why the scenario covers it."}, + {"statement": "Second candidate statement.", "correct": false, "rationale": "Why the patch contradicts or does not introduce it."}, + {"statement": "Third candidate statement.", "correct": false, "rationale": "Why this is false for the selected change."}, + {"statement": "Fourth candidate statement.", "correct": false, "rationale": "Why this is false for the selected change."} + ] + } + ``` + + This is a schema example, not ready-to-use question content. Author four + concrete, similarly specific statements with exactly one supported by the + patch. Avoid bundled claims outside the scenario, trivia, obvious nonsense, + uniquely repeated keywords, or an answer longer than all the distractors. + A clarification-only change is valid; do not claim it changes the outcome. + Check each rationale against the actual patch, independently of the report. +5. **Freeze all cases.** `freeze` validates inputs, shuffles options, writes + the private answer key and public questions, and records input hashes. + After freezing, do not edit fixtures, questions, or the key. Start a new + evaluation if the design must change; retain the abandoned session and why. + +Scenario design is model-assisted maintainer work, not a claim that the installed +Behavior Diff plugin independently drafted the scenario. When delegating, keep +scenario workers separate from the quiz options and final answer key. + +## Consent and live execution + +Request fresh approval for 30 Claude Code trial executions plus five extraction +calls and any additional model-assisted preparation. Explain that private skill +text goes to the configured provider. Record the approval in the tracking issue; +never assume a previous session's approval applies. `--approve-live` is the +operator's attestation of that approval, not a substitute for asking. + +The helper launches the unchanged local `behavior-diff.sh`, selecting Claude +`opus` for trials and Claude `sonnet` for extraction. A local wrapper disables +user customizations, hooks, MCP, commands, writes, and network tools. This is a +read-only decision/output replay, not a production integration test. Either +Claude Code or Codex can orchestrate this skill, but this workflow intentionally +uses the same Claude trial stack; do not silently substitute another host. + +Run all cases sequentially with `run SESSION --approve-live`, or dispatch +separate cases with `--case N` after freezing. Each case has its own output root. +The normal runner executes six trial processes per case. Do not launch many +cases concurrently without considering the provider's rate limits. + +The runner's automatic opening of the full report is suppressed until the +human submits the quiz; use only the quiz URL before submission. + +An attempt is recorded before launching. Do not rerun until a favorable result +appears. Preserve failures and missing extraction. If the runner cannot produce +a report, report the incomplete case and its saved diagnostics; do not substitute +a synthetic report or another commit. Never invoke a live runner from CI. + +Check `runner.log`, run evidence, and the original trial traces. Verify exact +Before/After target bytes, nonempty results, target reads, actual model identity +where retained, and tool restrictions. A REVIEW grade is neutral, not a failure. +Do not represent an answer describing an action as evidence of execution. + +## Quiz and scoring + +`build` creates blinded excerpts from the original saved report HTML. It retains +the generated headline, evidence, Before/After cards, meaning, Other findings, +and limits. It removes instruction intent/diff, full scenario/expected behavior, +commit metadata, and links to unblinded evidence. It must not rewrite report +prose or synthesize missing summaries. The projection fails if the report shape +is no longer recognized; update it deliberately alongside a report redesign. + +`serve SESSION --port 0` prints an available loopback URL. Only public quiz assets +are served before submission. The answer key stays on the server. Check desktop +and narrow browser views, including expanded findings, without answering the +human's quiz. Test submissions belong in a separate synthetic session. + +The human chooses one of four options, supplies confidence, and can flag +insufficient evidence or add a note. The first complete submission is saved; +refreshing or submitting again does not replace it. Correct answers, rationales, +source commit links, and the full reports become available afterward. Keep the +service running until the human finishes; restart `serve` to resume later. + +`results SESSION` reads the saved submission without a model call. Report the +score out of five, confidence, insufficient-evidence count, and comments. +Investigate misses against the original reports, scenarios, patches, and traces. +Do not infer comprehension from keyword matching or score alone. Inspect factual +consistency even in correctly answered cases. Preserve the first score; do not +rescore after changing a question or revealing the answers. + +A random guess averages 1.25/5. Five possibly correlated cases cannot establish +product-wide accuracy. Separate summary readability/fidelity from scenario +coverage and quiz ambiguity. An absence of observable difference is useful +evidence. Later wording comparisons need fresh cases or an independent reader +because this human now knows these answers. diff --git a/.agents/skills/run-behavior-diff-human-evaluation/scripts/evaluate.py b/.agents/skills/run-behavior-diff-human-evaluation/scripts/evaluate.py new file mode 100755 index 0000000..1f7ecad --- /dev/null +++ b/.agents/skills/run-behavior-diff-human-evaluation/scripts/evaluate.py @@ -0,0 +1,1420 @@ +#!/usr/bin/env python3 +"""Private, manual, fixed-source Behavior Diff human-evaluation lifecycle.""" + +import argparse +import datetime +import hashlib +import io +import importlib.util +import json +import os +from pathlib import Path, PurePosixPath +import random +import re +import secrets +import shutil +import stat +import subprocess +import sys +import tarfile +import uuid + +REPO_ROOT = Path(__file__).resolve().parents[4] +SOURCE_REPO = "DataRecce/recce-team" +SOURCE_URL = "https://github.com/DataRecce/recce-team.git" +CASE_COUNT = 5 +RECEIPT = "freeze-receipt.json" +SHA = re.compile(r"[0-9a-f]{40}\Z") +CI_MARKERS = ( + "CI", + "GITHUB_ACTIONS", + "GITLAB_CI", + "BUILDKITE", + "CIRCLECI", + "TRAVIS", + "TF_BUILD", + "JENKINS_URL", + "TEAMCITY_VERSION", +) + + +class EvaluationError(ValueError): + pass + + +def require(condition, message): + if not condition: + raise EvaluationError(message) + + +def now(): + return datetime.datetime.now(datetime.timezone.utc).isoformat() + + +def digest(data): + return hashlib.sha256(data).hexdigest() + + +def save_json(path, value, exclusive=False): + path = Path(path) + with path.open("x" if exclusive else "w", encoding="utf-8") as stream: + os.chmod(path, 0o600) + json.dump(value, stream, indent=2, ensure_ascii=False) + stream.write("\n") + + +def read_json(path): + try: + return json.loads(Path(path).read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + raise EvaluationError("Cannot read {}: {}".format(path, exc)) from exc + + +def within(path, root): + try: + path.relative_to(root) + return True + except ValueError: + return False + + +def no_symlinks(path): + path = Path(os.path.abspath(str(path.expanduser()))) + for component in (path,) + tuple(path.parents): + require( + not component.is_symlink(), "Symlink path is unsafe: {}".format(component) + ) + return path + + +def relative_file(value): + require( + isinstance(value, str) and value and "\x00" not in value and "\\" not in value, + "Expected a safe relative file path", + ) + path = PurePosixPath(value) + require( + not path.is_absolute() + and all(part not in (".", "..", ".git") for part in value.split("/")), + "Unsafe relative path: {}".format(value), + ) + require(str(path) == value, "Noncanonical relative path: {}".format(value)) + return path + + +def private_root(session, must_exist=True): + session = no_symlinks(Path(session)) + require( + not within(session, REPO_ROOT.resolve()) + and not within(REPO_ROOT.resolve(), session), + "Evaluation state must be outside the code repository", + ) + if must_exist: + require( + session.is_dir(), "Session directory does not exist: {}".format(session) + ) + mode = session.stat().st_mode + require(mode & 0o077 == 0, "Session root must be private (chmod 700)") + require( + session.stat().st_uid == os.getuid(), + "Session root must belong to this user", + ) + files(session, include_git=True) + return session + + +def files(root, include_git=False, skip_caches=False): + """Return regular files, rejecting links and special files, including empty dirs' safety.""" + result = [] + root = Path(root) + require( + root.is_dir() and not root.is_symlink(), "Unsafe directory: {}".format(root) + ) + for directory, dirs, names in os.walk(root, followlinks=False): + for name in list(dirs): + child = Path(directory) / name + require(not child.is_symlink(), "Symlink is unsafe: {}".format(child)) + if (not include_git and name == ".git") or ( + skip_caches and name == "__pycache__" + ): + dirs.remove(name) + for name in names: + path = Path(directory) / name + require( + not path.is_symlink() and stat.S_ISREG(path.stat().st_mode), + "Nonregular file is unsafe: {}".format(path), + ) + if skip_caches and path.suffix in (".pyc", ".pyo"): + continue + result.append(path) + return sorted(result) + + +def git_environment(): + env = os.environ.copy() + for key in list(env): + if key.startswith("GIT_"): + env.pop(key) + env.update( + { + "GIT_CONFIG_GLOBAL": os.devnull, + "GIT_CONFIG_NOSYSTEM": "1", + "GIT_TERMINAL_PROMPT": "0", + "GIT_CONFIG_COUNT": "2", + "GIT_CONFIG_KEY_0": "core.hooksPath", + "GIT_CONFIG_VALUE_0": os.devnull, + "GIT_CONFIG_KEY_1": "commit.gpgSign", + "GIT_CONFIG_VALUE_1": "false", + } + ) + return env + + +def git(repo, *arguments, allow_failure=False): + proc = subprocess.run( + ["git", "-C", str(repo), *arguments], capture_output=True, env=git_environment() + ) + if not allow_failure and proc.returncode: + raise EvaluationError( + "git {} failed: {}".format( + " ".join(arguments), proc.stderr.decode("utf-8", "replace").strip() + ) + ) + return proc + + +def git_text(repo, *arguments): + return git(repo, *arguments).stdout.decode("utf-8").strip() + + +def code_fingerprint(root=None): + root = Path(root or REPO_ROOT).resolve() + inventory = {} + targets = ( + root / "plugin", + root / "bin/behavior-diff", + root / ".agents/skills/run-behavior-diff-human-evaluation/scripts", + root / ".agents/skills/run-behavior-diff-human-evaluation/assets", + ) + for target in targets: + require(not target.is_symlink(), "Code symlink is unsafe: {}".format(target)) + if target.is_dir(): + entries = files(target, skip_caches=True) + elif target.is_file(): + entries = [target] + else: + continue + for entry in entries: + inventory[entry.relative_to(root).as_posix()] = { + "sha256": digest(entry.read_bytes()), + "executable": bool(entry.stat().st_mode & 0o111), + } + require(inventory, "No local Behavior Diff code found") + head = git_text(root, "rev-parse", "HEAD") + fingerprint = digest( + json.dumps({"head": head, "files": inventory}, sort_keys=True).encode() + ) + return { + "root": str(root), + "head": head, + "fingerprint": fingerprint, + "files": inventory, + } + + +def changed_files(repo, before, after): + raw = git( + repo, + "diff-tree", + "--no-commit-id", + "--name-status", + "-r", + "-z", + "--no-renames", + before, + after, + ).stdout.split(b"\0") + require(raw[-1] == b"", "Malformed git diff-tree output") + raw.pop() + require(len(raw) % 2 == 0, "Malformed git name/status pairs") + return [ + {"status": raw[i].decode("ascii"), "path": raw[i + 1].decode("utf-8")} + for i in range(0, len(raw), 2) + ] + + +def regular_blob(repo, commit, path): + entry = git(repo, "ls-tree", "-z", commit, "--", path).stdout + require( + entry.endswith(b"\0") and entry.count(b"\0") == 1, + "Missing or ambiguous source file: {}".format(path), + ) + metadata, found = entry[:-1].split(b"\t", 1) + mode, kind, blob = metadata.decode("ascii").split() + require( + found.decode("utf-8") == path + and mode in ("100644", "100755") + and kind == "blob", + "Source skill must be an existing regular file: {}".format(path), + ) + return blob + + +def substantive_change(repo, before, after, path): + statistics = git( + repo, + "diff", + "--no-ext-diff", + "--no-textconv", + "--ignore-all-space", + "--ignore-blank-lines", + "--numstat", + "-z", + before, + after, + "--", + path, + ).stdout + return bool(statistics) and statistics.split(b"\t", 2)[:2] != [b"0", b"0"] + + +def enumerate_candidates(repo, tip): + candidates, excluded = [], [] + rows = git_text(repo, "rev-list", "--parents", tip).splitlines() + for row in rows: + sha, *parents = row.split() + item = {"sha": sha, "parents": parents} + if len(parents) != 1: + item["reason"] = "root" if not parents else "merge" + excluded.append(item) + continue + before = parents[0] + changes = changed_files(repo, before, sha) + item.update({"before_sha": before, "changes": changes}) + skills = [ + change + for change in changes + if PurePosixPath(change["path"]).name == "SKILL.md" + ] + reason = None + if len(skills) != 1 or skills[0]["status"] != "M": + reason = "not-exactly-one-existing-modified-skill" + else: + path = skills[0]["path"] + try: + relative_file(path) + before_blob = regular_blob(repo, before, path) + after_blob = regular_blob(repo, sha, path) + except EvaluationError: + reason = "unsafe-or-nonregular-skill" + directory = str(PurePosixPath(path).parent) + prefix = "" if directory == "." else directory + "/" + if reason is None and any( + change["path"] != path and change["path"].startswith(prefix) + for change in changes + ): + reason = "companion-change-in-skill-directory" + if reason is None and not substantive_change(repo, before, sha, path): + reason = "whitespace-only" + if reason is None: + item.update( + { + "skill_path": path, + "before_blob": before_blob, + "after_blob": after_blob, + } + ) + candidates.append(item) + if reason: + item["reason"] = reason + excluded.append(item) + return { + "reachable_commits": len(rows), + "candidates": candidates, + "excluded": excluded, + } + + +def case_entry(candidate, case_id): + return { + "id": case_id, + "sha": candidate["sha"], + "before_sha": candidate["before_sha"], + "skill_path": candidate["skill_path"], + } + + +def snapshot_case(session, entry): + case = session / "case-{}".format(entry["id"]) + case.mkdir(mode=0o700) + source = session / "source" + for filename, commit in ( + ("before.md", entry["before_sha"]), + ("after.md", entry["sha"]), + ): + path = case / filename + path.write_bytes( + git(source, "show", "{}:{}".format(commit, entry["skill_path"])).stdout + ) + path.chmod(0o600) + patch = case / "patch.diff" + patch.write_bytes( + git( + source, + "diff", + "--binary", + "--no-ext-diff", + "--no-textconv", + entry["before_sha"], + entry["sha"], + "--", + entry["skill_path"], + ).stdout + ) + patch.chmod(0o600) + + +def populate_session(session, source_repo=None, seed=None): + """Populate a new private session. Local source/seed injection is test-only API, not CLI.""" + session = private_root(session, must_exist=False) + session.mkdir(mode=0o700, parents=True, exist_ok=False) + try: + command = [ + "git", + "clone", + "--no-checkout", + "--no-local", + str(source_repo) if source_repo is not None else SOURCE_URL, + str(session / "source"), + ] + clone_environment = git_environment() + if source_repo is None: + require(shutil.which("gh"), "Authenticated GitHub CLI (gh) is required") + # Local Git configuration stays isolated, but private upstream access + # still needs an explicit credential provider. Never capture its token. + clone_environment.update( + { + "GIT_CONFIG_COUNT": "3", + "GIT_CONFIG_KEY_2": "credential.helper", + "GIT_CONFIG_VALUE_2": "!gh auth git-credential", + } + ) + proc = subprocess.run(command, capture_output=True, env=clone_environment) + (session / "clone.log").write_bytes(proc.stdout + proc.stderr) + (session / "clone.log").chmod(0o600) + require( + proc.returncode == 0, + "Fixed upstream clone failed; see {}".format(session / "clone.log"), + ) + source = session / "source" + tip = git_text(source, "rev-parse", "refs/remotes/origin/main") + require(SHA.fullmatch(tip), "Upstream main must resolve to a full SHA") + git(source, "update-ref", "refs/heads/human-evaluation-source", tip) + sampling = enumerate_candidates(source, tip) + seed = secrets.randbits(128) if seed is None else seed + require( + type(seed) is int and seed >= 0, + "Sampling seed must be a nonnegative integer", + ) + order = list(sampling["candidates"]) + random.Random(seed).shuffle(order) + require( + len(order) >= CASE_COUNT, + "Fewer than five eligible commits; no substitutions allowed", + ) + sampling.update( + { + "source_repo": SOURCE_REPO, + "source_tip": tip, + "seed": seed, + "algorithm": "uniform-random.Random-shuffle-v1", + "order": [item["sha"] for item in order], + } + ) + manifest = { + "schema_version": 1, + "id": session.name, + "created_at": now(), + "source_repo": SOURCE_REPO, + "source_tip": tip, + "seed": seed, + "code": code_fingerprint(), + "trials": 3, + "trial_model": "opus", + "extract_model": "sonnet", + "cases": [ + case_entry(candidate, index + 1) + for index, candidate in enumerate(order[:CASE_COUNT]) + ], + "next_candidate": CASE_COUNT, + "rejections": [], + } + save_json(session / "sampling.json", sampling, exclusive=True) + for entry in manifest["cases"]: + snapshot_case(session, entry) + save_json(session / "session.json", manifest, exclusive=True) + return manifest + except Exception: + # Keep the failed clone/sampling evidence in its private unique directory. + raise + + +def init_session(): + base = private_root( + Path.home() / ".behavior-diff/human-evaluations", must_exist=False + ) + base.mkdir(mode=0o700, parents=True, exist_ok=True) + identifier = ( + datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%dT%H%M%SZ-") + + uuid.uuid4().hex + ) + session = base / identifier + populate_session(session) + return session + + +def load_session(session): + session = private_root(session) + manifest = read_json(session / "session.json") + require( + isinstance(manifest, dict) and manifest.get("schema_version") == 1, + "Unsupported session manifest", + ) + require( + manifest.get("id") == session.name + and manifest.get("source_repo") == SOURCE_REPO, + "Manifest must name this session and fixed DataRecce/recce-team source", + ) + require( + isinstance(manifest.get("source_tip"), str) + and SHA.fullmatch(manifest["source_tip"]), + "Invalid pinned source tip", + ) + require( + type(manifest.get("seed")) is int and manifest["seed"] >= 0, + "Invalid sampling seed", + ) + require( + manifest.get("trials") == 3 + and manifest.get("trial_model") == "opus" + and manifest.get("extract_model") == "sonnet", + "This protocol fixes three trials and Claude Opus/Sonnet", + ) + code = manifest.get("code") + require( + isinstance(code, dict) + and code.get("root") == str(REPO_ROOT.resolve()) + and isinstance(code.get("head"), str) + and isinstance(code.get("fingerprint"), str), + "Manifest must identify this local code checkout", + ) + cases = manifest.get("cases") + require( + isinstance(cases, list) and len(cases) == CASE_COUNT, + "Exactly five cases are required", + ) + seen = set() + for index, entry in enumerate(cases, 1): + require( + isinstance(entry, dict) + and type(entry.get("id")) is int + and entry["id"] == index, + "Case ids must be 1 through 5 in order", + ) + require( + all( + isinstance(entry.get(key), str) and SHA.fullmatch(entry[key]) + for key in ("sha", "before_sha") + ), + "Cases must pin full source SHAs", + ) + require(entry["sha"] not in seen, "Cases must be unique") + seen.add(entry["sha"]) + relative_file(entry.get("skill_path")) + require( + PurePosixPath(entry["skill_path"]).name == "SKILL.md", + "Only existing SKILL.md edits are eligible", + ) + require((session / "case-{}".format(index)).is_dir(), "Missing case directory") + source = session / "source" + require( + source.is_dir() and (source / ".git").is_dir(), "Missing private source clone" + ) + require( + git_text(source, "rev-parse", "refs/remotes/origin/main") + == manifest["source_tip"], + "Pinned upstream main changed", + ) + sampling = read_json(session / "sampling.json") + require( + isinstance(sampling, dict) + and sampling.get("source_repo") == SOURCE_REPO + and sampling.get("source_tip") == manifest["source_tip"] + and sampling.get("seed") == manifest["seed"], + "Sampling provenance does not match manifest", + ) + entries = sampling.get("candidates") + require( + isinstance(entries, list) + and all( + isinstance(item, dict) + and all( + isinstance(item.get(key), str) + for key in ("sha", "before_sha", "skill_path") + ) + for item in entries + ), + "Invalid sampling candidates", + ) + candidates = {candidate["sha"]: candidate for candidate in entries} + require(len(candidates) == len(entries), "Duplicate sampled candidates") + order = sampling.get("order") + require( + isinstance(order, list) + and all(isinstance(item, str) for item in order) + and len(order) == len(candidates) + and set(order) == set(candidates), + "Sampling order must cover every eligible candidate once", + ) + expected = list(candidates.values()) + random.Random(manifest["seed"]).shuffle(expected) + require(order == [item["sha"] for item in expected], "Sampling shuffle changed") + require( + type(manifest.get("next_candidate")) is int + and CASE_COUNT <= manifest["next_candidate"] <= len(order), + "Invalid replacement cursor", + ) + require( + isinstance(manifest.get("rejections"), list), "Invalid rejection provenance" + ) + selected = [ + case_entry(candidates[sha], index + 1) + for index, sha in enumerate(order[:CASE_COUNT]) + ] + for cursor, rejection in enumerate(manifest["rejections"], CASE_COUNT): + require( + isinstance(rejection, dict) + and type(rejection.get("case")) is int + and 1 <= rejection["case"] <= CASE_COUNT + and isinstance(rejection.get("reason"), str) + and rejection["reason"].strip() + and cursor < len(order), + "Invalid replacement record", + ) + index = rejection["case"] - 1 + replacement = case_entry(candidates[order[cursor]], index + 1) + require( + rejection.get("rejected") == selected[index] + and rejection.get("replacement") == replacement, + "Replacement must use the next unused random candidate", + ) + selected[index] = replacement + require( + manifest["next_candidate"] == CASE_COUNT + len(manifest["rejections"]) + and cases == selected, + "Case selection no longer follows the recorded random order", + ) + return manifest + + +def verify_sampling(session, manifest): + sampling = read_json(session / "sampling.json") + actual = enumerate_candidates(session / "source", manifest["source_tip"]) + require( + all(sampling.get(key) == value for key, value in actual.items()), + "Complete sampling candidates/exclusions no longer match pinned upstream history", + ) + + +def replace_case(session, case_id, reason): + session = private_root(session) + manifest = load_session(session) + require( + not (session / RECEIPT).exists() and not (session / "answer-key.json").exists(), + "Replacement is forbidden after freeze", + ) + require( + not any( + (session / "case-{}".format(index) / "attempt.json").exists() + for index in range(1, 6) + ), + "Replacement is forbidden after any live attempt", + ) + require( + type(case_id) is int + and 1 <= case_id <= CASE_COUNT + and isinstance(reason, str) + and reason.strip(), + "Replacement requires a case 1..5 and a nonempty reason", + ) + case = session / "case-{}".format(case_id) + require( + {path.name for path in case.iterdir()} + == {"before.md", "after.md", "patch.diff"}, + "Case has authored artifacts; preserve them rather than replacing", + ) + verify_sampling(session, manifest) + old = manifest["cases"][case_id - 1] + verify_source_case(session, old) + sampling = read_json(session / "sampling.json") + cursor = manifest["next_candidate"] + require(cursor < len(sampling["order"]), "No unused eligible candidate remains") + candidates = {item["sha"]: item for item in sampling["candidates"]} + replacement = case_entry(candidates[sampling["order"][cursor]], case_id) + archive = session / "rejected" + archive.mkdir(mode=0o700, exist_ok=True) + case.rename(archive / "case-{}-{}".format(case_id, old["sha"])) + snapshot_case(session, replacement) + manifest["cases"][case_id - 1] = replacement + manifest["next_candidate"] = cursor + 1 + manifest["rejections"].append( + { + "case": case_id, + "rejected": old, + "replacement": replacement, + "reason": reason.strip(), + "at": now(), + } + ) + save_json(session / "session.json", manifest) + return replacement + + +def verify_source_case(session, entry): + source = session / "source" + tip = read_json(session / "session.json")["source_tip"] + require( + git( + source, "merge-base", "--is-ancestor", entry["sha"], tip, allow_failure=True + ).returncode + == 0, + "Source case is not reachable from pinned main", + ) + require( + git_text(source, "rev-list", "--parents", "-n", "1", entry["sha"]).split() + == [entry["sha"], entry["before_sha"]], + "Source case is not a nonmerge parent-child pair", + ) + changes = changed_files(source, entry["before_sha"], entry["sha"]) + path = entry["skill_path"] + skill_changes = [ + item for item in changes if PurePosixPath(item["path"]).name == "SKILL.md" + ] + require( + skill_changes == [{"status": "M", "path": path}], + "Source case is not one existing SKILL.md edit", + ) + prefix = ( + "" + if str(PurePosixPath(path).parent) == "." + else str(PurePosixPath(path).parent) + "/" + ) + require( + not any( + item["path"] != path and item["path"].startswith(prefix) for item in changes + ), + "Source case includes companion changes in the skill directory", + ) + case = session / "case-{}".format(entry["id"]) + for filename, commit in ( + ("before.md", entry["before_sha"]), + ("after.md", entry["sha"]), + ): + regular_blob(source, commit, path) + require( + (case / filename).read_bytes() + == git(source, "show", "{}:{}".format(commit, path)).stdout, + "Source snapshot bytes changed: {}".format(case / filename), + ) + expected = git( + source, + "diff", + "--binary", + "--no-ext-diff", + "--no-textconv", + entry["before_sha"], + entry["sha"], + "--", + path, + ).stdout + require( + (case / "patch.diff").read_bytes() == expected, "Source patch bytes changed" + ) + require( + substantive_change(source, entry["before_sha"], entry["sha"], path), + "Source change is whitespace-only or unreadable", + ) + + +def validate_case(session, entry): + verify_source_case(session, entry) + case = session / "case-{}".format(entry["id"]) + scenario = read_json(case / "scenario.json") + require(isinstance(scenario, dict), "Scenario must be an object") + target = str(relative_file(scenario.get("file"))) + require( + PurePosixPath(target).name == "SKILL.md", "Fixture target must remain SKILL.md" + ) + require( + isinstance(scenario.get("task"), str) and scenario["task"].strip(), + "Scenario task must be nonempty", + ) + fixture = case / "fixture" + files(fixture) + require( + (fixture / ".git").is_dir() and not (fixture / ".git").is_symlink(), + "Fixture must have its own committed Before git repository", + ) + require( + git_text(fixture, "rev-parse", "--show-toplevel") == str(fixture), + "Fixture git root escaped its directory", + ) + alternates = fixture / ".git/objects/info/alternates" + require( + not alternates.exists() or not alternates.read_bytes().strip(), + "Fixture may not use external object alternates", + ) + require( + all( + not row.startswith(b"160000 ") + for row in git(fixture, "ls-files", "--stage").stdout.splitlines() + ), + "Fixture submodules are not supported", + ) + regular_blob(fixture, "HEAD", target) + before = git(fixture, "show", "HEAD:" + target).stdout + require( + before == (case / "before.md").read_bytes(), + "Fixture HEAD target must contain exact Before bytes", + ) + with tarfile.open( + fileobj=io.BytesIO(git(fixture, "archive", "--format=tar", "HEAD").stdout) + ) as archive: + try: + member = archive.getmember(target) + except KeyError as exc: + raise EvaluationError( + "Fixture archive omits target (export-ignore is unsafe)" + ) from exc + require( + member.isfile() and archive.extractfile(member).read() == before, + "Fixture archive must retain exact Before bytes (no export-subst)", + ) + require( + (fixture / target).is_file() + and (fixture / target).read_bytes() == (case / "after.md").read_bytes(), + "Fixture working target must contain exact After bytes", + ) + changes = git(fixture, "diff", "--name-status", "-z", "--no-renames", "HEAD").stdout + require( + changes == b"M\0" + target.encode("utf-8") + b"\0", + "After must be the sole working change", + ) + status = git( + fixture, "status", "--porcelain=v1", "-z", "--untracked-files=all", "--ignored" + ).stdout + require( + status + in ( + b" M " + target.encode() + b"\0", + b"M " + target.encode() + b"\0", + b"MM " + target.encode() + b"\0", + ), + "Fixture contains untracked, ignored, or extra changes", + ) + return { + "file": target, + "task": scenario["task"], + "head": git_text(fixture, "rev-parse", "HEAD"), + } + + +def frozen_inputs(session, manifest): + inventory = {} + fixed = [ + session / name + for name in ( + "session.json", + "sampling.json", + "answer-key.json", + "public/questions.json", + ) + ] + heads = {} + for entry in manifest["cases"]: + case = session / "case-{}".format(entry["id"]) + fixed.extend( + case / name + for name in ( + "before.md", + "after.md", + "patch.diff", + "scenario.json", + "question.json", + ) + ) + fixed.extend(files(case / "fixture")) + heads[str(entry["id"])] = git_text(case / "fixture", "rev-parse", "HEAD") + for path in fixed: + require( + path.is_file() and not path.is_symlink(), + "Missing frozen input: {}".format(path), + ) + inventory[path.relative_to(session).as_posix()] = { + "sha256": digest(path.read_bytes()), + "executable": bool(path.stat().st_mode & 0o111), + } + return {"files": inventory, "fixture_heads": heads} + + +def quiz_module(): + path = Path(__file__).with_name("quiz.py") + spec = importlib.util.spec_from_file_location("behavior_diff_human_quiz", path) + module = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = module + spec.loader.exec_module(module) + return module + + +def freeze_session(session): + session = private_root(session) + manifest = load_session(session) + require( + not (session / RECEIPT).exists() + and not (session / "answer-key.json").exists() + and not (session / "public/questions.json").exists(), + "Freeze is one-time and cannot overwrite questions/key", + ) + require( + not any( + (session / "case-{}".format(i) / "attempt.json").exists() + for i in range(1, 6) + ), + "Cannot freeze after a live attempt", + ) + current = code_fingerprint() + require( + current == manifest["code"], + "Local code changed since init; create a fresh session", + ) + verify_sampling(session, manifest) + for entry in manifest["cases"]: + validate_case(session, entry) + quiz_module().freeze_questions(session, manifest) + receipt = { + "schema_version": 1, + "session_id": manifest["id"], + "frozen_at": now(), + "code": current, + "inputs": frozen_inputs(session, manifest), + } + save_json(session / RECEIPT, receipt, exclusive=True) + return receipt + + +def verify_frozen(session, manifest=None, check_code=True): + session = private_root(session) + manifest = load_session(session) if manifest is None else manifest + receipt = read_json(session / RECEIPT) + require( + isinstance(receipt, dict) + and receipt.get("schema_version") == 1 + and receipt.get("session_id") == manifest["id"] + and receipt.get("code") == manifest["code"], + "Invalid frozen receipt", + ) + require( + receipt.get("inputs") == frozen_inputs(session, manifest), + "Frozen inputs drifted; never run edited cases", + ) + for entry in manifest["cases"]: + validate_case(session, entry) + if check_code: + require( + code_fingerprint() == receipt["code"], + "Local code/HEAD changed after freeze; create a fresh session", + ) + + +LAUNCHER = """#!/usr/bin/env python3 +import json, os, pathlib, subprocess, sys, threading, uuid +REAL = __REAL__ +LOGS = pathlib.Path(__LOGS__) +SAFETY = ['--safe-mode', '--restricted', '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}', + '--setting-sources', '', '--settings', '{"disableAllHooks":true}', '--no-chrome', + '--disable-slash-commands', '--tools', 'Read,Grep,Glob', '--permission-mode', 'dontAsk', + '--no-session-persistence'] +os.umask(0o077) +call = LOGS / uuid.uuid4().hex +call.mkdir(mode=0o700) +command = [REAL] + sys.argv[1:] + SAFETY +(call / 'invocation.json').write_text(json.dumps({'command': command, 'cwd': os.getcwd(), + 'effective_tools': ['Read', 'Grep', 'Glob']})) +proc = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE) +def copy(source, destination, filename): + with (call / filename).open('wb') as log: + while True: + data = source.read1(65536) + if not data: + break + log.write(data) + log.flush() + destination.write(data) + destination.flush() +threads = [threading.Thread(target=copy, args=(proc.stdout, sys.stdout.buffer, 'stdout.log')), + threading.Thread(target=copy, args=(proc.stderr, sys.stderr.buffer, 'stderr.log'))] +for thread in threads: + thread.start() +status = proc.wait() +for thread in threads: + thread.join() +(call / 'exit.json').write_text(json.dumps({'returncode': status})) +sys.exit(status) +""" + + +def prepare_launcher(session): + # Resolve before adding the private bin to PATH: never recursively invoke our wrapper. + real = shutil.which("claude") + require( + real is not None, "Real Claude executable is required for approved live runs" + ) + real = Path(real).resolve() + require( + real.is_file() and os.access(real, os.X_OK) and not within(real, session), + "Claude must resolve to a real executable outside this session", + ) + require( + shutil.which("jq") is not None, "jq is required by the unchanged local runner" + ) + binary = session / "bin" + binary.mkdir(mode=0o700, exist_ok=True) + logs = session / "launcher-logs" + logs.mkdir(mode=0o700, exist_ok=True) + content = LAUNCHER.replace("__REAL__", repr(str(real))).replace( + "__LOGS__", repr(str(logs)) + ) + deny_open = "#!/bin/sh\nprintf '%s\\n' 'Full reports are gated until quiz submission.' >&2\nexit 1\n" + for name, body in (("claude", content), ("open", deny_open)): + launcher = binary / name + if launcher.exists(): + require( + launcher.read_text() == body, + "Existing {} launcher differs from the enforced guard".format(name), + ) + else: + with launcher.open("x") as stream: + stream.write(body) + launcher.chmod(0o700) + return { + "real_executable": str(real), + "launcher": str(binary / "claude"), + "sha256": digest(content.encode()), + "open_denial_sha256": digest(deny_open.encode()), + "tools": ["Read", "Grep", "Glob"], + "trials": 3, + "trial_model": "opus", + "extract_model": "sonnet", + } + + +def trace_evidence(trace, project, target): + events, malformed = [], 0 + if trace.is_file(): + for line in trace.read_text(encoding="utf-8", errors="replace").splitlines(): + try: + value = json.loads(line) + if isinstance(value, dict): + events.append(value) + else: + malformed += 1 + except ValueError: + malformed += 1 + models, advertised_tools, used_tools, reads, responses, costs = ( + set(), + set(), + [], + [], + {}, + [], + ) + final = [] + for event in events: + if isinstance(event.get("model"), str): + models.add(event["model"]) + for tool in ( + event.get("tools", []) if isinstance(event.get("tools"), list) else [] + ): + if isinstance(tool, str): + advertised_tools.add(tool) + message = event.get("message", {}) + if not isinstance(message, dict): + continue + if isinstance(message.get("model"), str): + models.add(message["model"]) + content = message.get("content", []) + if not isinstance(content, list): + content = [] + for block in content: + if not isinstance(block, dict): + continue + if block.get("type") == "tool_use": + used_tools.append(block.get("name")) + if block.get("name") == "Read": + arguments = block.get("input", {}) + path = ( + arguments.get("file_path", "") + if isinstance(arguments, dict) + else "" + ) + reads.append({"id": block.get("id"), "path": path}) + elif block.get("type") == "tool_result": + responses[block.get("tool_use_id")] = block + if event.get("type") == "result": + final.append(event) + if isinstance(event.get("total_cost_usd"), (int, float)): + costs.append(event["total_cost_usd"]) + successful, missing, failed, unmatched, target_success = 0, 0, 0, 0, 0 + for read in reads: + response = responses.get(read["id"]) + if response is None: + read["outcome"] = "unmatched" + unmatched += 1 + elif response.get("is_error"): + text = json.dumps(response.get("content", "")).lower() + absent = any( + term in text for term in ("does not exist", "no such file", "not found") + ) + read["outcome"] = "missing" if absent else "failed" + missing += int(absent) + failed += int(not absent) + else: + read["outcome"] = "success" + successful += 1 + if isinstance(read["path"], str): + path = Path(read["path"]) + if not path.is_absolute(): + path = project / path + if os.path.normpath(str(path)) == str(project / target): + target_success += 1 + success = any( + event.get("result") + and not event.get("is_error") + and event.get("subtype", "success") == "success" + for event in final + ) + return { + "actual_models": sorted(models), + "advertised_tools": sorted(advertised_tools), + "used_tools": used_tools, + "success": bool(success), + "malformed_events": malformed, + "source_reads": { + "successful": successful, + "missing": missing, + "failed": failed, + "unmatched": unmatched, + "successful_target": target_success, + "calls": reads, + }, + "cost_usd": sum(costs) if costs else None, + "usage": [event.get("usage") for event in final if "usage" in event], + } + + +def launcher_evidence(session, case): + logs = session / "launcher-logs" + calls = [] + if not logs.is_dir(): + return calls + for call in sorted(logs.iterdir()): + invocation = read_json(call / "invocation.json") + cwd = Path(invocation["cwd"]) + if not within(cwd, case): + continue + command = invocation["command"] + model = command[command.index("--model") + 1] if "--model" in command else None + exit_path = call / "exit.json" + calls.append( + { + "path": str(call.relative_to(session)), + "cwd": str(cwd.relative_to(session)), + "requested_model": model, + "enforced_tools": invocation["effective_tools"], + "exit_code": read_json(exit_path)["returncode"] + if exit_path.exists() + else None, + "logs": { + name: digest((call / name).read_bytes()) + if (call / name).is_file() + else None + for name in ( + "invocation.json", + "stdout.log", + "stderr.log", + "exit.json", + ) + }, + } + ) + return calls + + +def collect_evidence(session, entry, scenario, exit_code, settings): + case = session / "case-{}".format(entry["id"]) + runs_root = case / "state/runs" + runs = ( + sorted(path for path in runs_root.iterdir() if path.is_dir()) + if runs_root.is_dir() + else [] + ) + evidence = { + "session_id": session.name, + "case": entry["id"], + "finished_at": now(), + "runner_exit_code": exit_code, + "settings": settings, + "runs": [path.name for path in runs], + "trials": [], + "artifacts": {}, + "blocked_reasons": [], + "launcher_calls": launcher_evidence(session, case), + } + if len(runs) != 1: + evidence["blocked_reasons"].append( + "expected exactly one report run, found {}".format(len(runs)) + ) + else: + run = runs[0] + targets = case / "trial-targets" + targets.mkdir(mode=0o700) + for variant in ("before", "after"): + expected = (case / (variant + ".md")).read_bytes() + for index in range(1, 4): + trial_id = "{}-{}".format(variant, index) + trial = run / trial_id + project = trial / "project" + item = trace_evidence(trial / "trace.jsonl", project, scenario["file"]) + item["id"] = trial_id + target = project / scenario["file"] + item["target_present"] = target.is_file() and not target.is_symlink() + if item["target_present"]: + data = target.read_bytes() + saved = targets / (trial_id + ".md") + saved.write_bytes(data) + saved.chmod(0o600) + item.update( + { + "target_sha256": digest(data), + "target_matches_frozen": data == expected, + } + ) + else: + item["target_matches_frozen"] = False + if ( + not item["success"] + or not item["target_matches_frozen"] + or item["malformed_events"] + or not item["source_reads"]["successful_target"] + or any( + tool not in ("Read", "Grep", "Glob") + for tool in item["used_tools"] + ) + ): + evidence["blocked_reasons"].append( + "{}: missing/invalid result, target bytes/read, or unsafe tools".format( + trial_id + ) + ) + evidence["trials"].append(item) + for filename in ( + "report.html", + "report.md", + "report-data.json", + "decisions.json", + ): + path = run / filename + evidence["artifacts"][filename] = ( + { + "path": str(path.relative_to(session)), + "sha256": digest(path.read_bytes()), + } + if path.is_file() + else None + ) + evidence["blocked"] = bool(evidence["blocked_reasons"]) + costs = [ + item["cost_usd"] for item in evidence["trials"] if item["cost_usd"] is not None + ] + evidence["cost_usd"] = sum(costs) if costs else None + evidence["cost_scope"] = ( + "Retained trial result events only; extraction cost unavailable unless retained in launcher logs." + ) + save_json(case / "run-evidence.json", evidence, exclusive=True) + return evidence + + +def run_session(session, approve_live=False, case_id=None): + session = private_root(session) + manifest = load_session(session) + require(approve_live is True, "Live model calls require explicit --approve-live") + require( + not any(marker in os.environ for marker in CI_MARKERS), + "Live human evaluations are forbidden in CI", + ) + verify_frozen(session, manifest) + require( + case_id is None or type(case_id) is int and 1 <= case_id <= CASE_COUNT, + "Case must be 1..5", + ) + cases = manifest["cases"] if case_id is None else [manifest["cases"][case_id - 1]] + # Refuse the whole requested batch before any calls if one case already has an attempt. + for entry in cases: + case = session / "case-{}".format(entry["id"]) + require( + not (case / "attempt.json").exists(), + "Case {} was already attempted; never retry".format(entry["id"]), + ) + require( + not (case / "state").exists() and not (case / "trial-targets").exists(), + "Case contains prior live state; never overwrite it", + ) + settings = prepare_launcher(session) + env = git_environment() + env["PATH"] = str(session / "bin") + os.pathsep + env.get("PATH", "") + env["BEHAVIOR_DIFF_TRIAL"] = "1" + env.pop("CLAUDECODE", None) + runner = REPO_ROOT / "plugin/skills/behavior-diff/scripts/behavior-diff.sh" + failures = [] + completed = [] + for entry in cases: + verify_frozen(session, manifest) + case = session / "case-{}".format(entry["id"]) + scenario = validate_case(session, entry) + command = [ + "bash", + str(runner), + "--agent", + "claude", + "--trials", + "3", + "--model", + "opus", + "--extract-agent", + "claude", + "--extract-model", + "sonnet", + "--file", + scenario["file"], + "--task", + scenario["task"], + ] + save_json( + case / "attempt.json", + { + "started_at": now(), + "command": command, + "settings": settings, + "approved_live": True, + "frozen_receipt_sha256": digest((session / RECEIPT).read_bytes()), + }, + exclusive=True, + ) + (case / "state").mkdir(mode=0o700) + env["BEHAVIOR_DIFF_HOME"] = str(case / "state") + try: + with (case / "runner.log").open("xb") as log: + os.chmod(case / "runner.log", 0o600) + proc = subprocess.run( + command, + cwd=case / "fixture", + env=env, + stdout=log, + stderr=subprocess.STDOUT, + ) + status = proc.returncode + except OSError as exc: + status = None + save_json(case / "launch-error.json", {"error": str(exc)}, exclusive=True) + evidence = collect_evidence(session, entry, scenario, status, settings) + completed.append(evidence) + if any( + evidence["artifacts"].get(name) is None + for name in ("report.html", "report.md", "report-data.json") + ): + failures.append( + "case {} has no complete report; retained evidence at {}".format( + entry["id"], case + ) + ) + require(not failures, "; ".join(failures)) + return completed + + +def parser(): + cli = argparse.ArgumentParser(description=__doc__) + commands = cli.add_subparsers(dest="command", required=True) + commands.add_parser( + "init", help="Create a fresh private session from fixed upstream main" + ) + for name in ("replace", "freeze", "run", "build", "serve", "results"): + command = commands.add_parser(name) + command.add_argument("session", type=Path) + if name == "replace": + command.add_argument("--case", type=int, required=True, choices=range(1, 6)) + command.add_argument("--reason", required=True) + elif name == "run": + command.add_argument("--approve-live", action="store_true") + command.add_argument("--case", type=int, choices=range(1, 6)) + elif name == "serve": + command.add_argument("--port", type=int, default=0) + return cli + + +def main(arguments=None): + args = parser().parse_args(arguments) + try: + if args.command == "init": + print(init_session()) + return 0 + session = private_root(args.session) + manifest = load_session(session) + if args.command == "replace": + print(json.dumps(replace_case(session, args.case, args.reason), indent=2)) + elif args.command == "freeze": + freeze_session(session) + print(str(session / RECEIPT)) + elif args.command == "run": + evidence = run_session(session, args.approve_live, args.case) + print( + json.dumps( + [ + { + "case": item["case"], + "blocked": item["blocked"], + "artifacts": item["artifacts"], + } + for item in evidence + ], + indent=2, + ) + ) + else: + verify_frozen(session, manifest, check_code=False) + quiz = quiz_module() + if args.command == "build": + quiz.build_quiz(session, REPO_ROOT) + print(str(session / "public/index.html")) + elif args.command == "serve": + require(0 <= args.port <= 65535, "Port must be 0..65535") + quiz.serve(session, args.port) + elif args.command == "results": + print(json.dumps(quiz.results(session), indent=2, ensure_ascii=False)) + return 0 + except (EvaluationError, OSError, ValueError) as exc: + print("human evaluation: {}".format(exc), file=sys.stderr) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py b/.agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py new file mode 100644 index 0000000..771b2c8 --- /dev/null +++ b/.agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py @@ -0,0 +1,980 @@ +"""Private, loopback-only human evaluation UI. No model calls or renderer imports.""" + +import hashlib +import html +from html.parser import HTMLParser +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import json +import os +from pathlib import Path +import random +import re +import tempfile +from datetime import datetime, timezone + + +ROOT = Path(__file__).resolve().parents[4] +ASSETS = Path(__file__).resolve().parent.parent / "assets" +LETTERS = "ABCD" +MAX_BODY = 32768 + + +def _json(path): + return json.loads(path.read_text(encoding="utf-8")) + + +def _bytes(value): + return (json.dumps(value, ensure_ascii=False, indent=2) + "\n").encode("utf-8") + + +def _hash(data): + return hashlib.sha256(data).hexdigest() + + +def _private_session(session): + session = Path(session).resolve() + if session == ROOT or ROOT in session.parents: + raise ValueError("Evaluation sessions must be outside the checkout.") + if not session.is_dir(): + raise ValueError("Session directory does not exist.") + return session + + +def _text(value, label, maximum): + if not isinstance(value, str) or not value.strip() or len(value) > maximum: + raise ValueError( + f"{label} must be nonempty text, at most {maximum} characters." + ) + if any(ord(char) < 32 and char not in "\n\t\r" for char in value): + raise ValueError(f"{label} contains control characters.") + return value + + +def _cases(manifest): + if ( + manifest.get("schema_version") != 1 + or manifest.get("source_repo") != "DataRecce/recce-team" + ): + raise ValueError("Unsupported evaluation session.") + _text(manifest.get("id"), "Session id", 200) + if type(manifest.get("seed")) is not int: + raise ValueError("Session seed must be an integer.") + cases = manifest.get("cases") + if not isinstance(cases, list) or len(cases) != 5: + raise ValueError("Exactly five cases are required.") + if any( + not isinstance(case, dict) or type(case.get("id")) is not int for case in cases + ): + raise ValueError("Case ids must be integers.") + if sorted(case["id"] for case in cases) != list(range(1, 6)): + raise ValueError("Case ids must be exactly 1 through 5.") + for case in cases: + for field in ("sha", "before_sha"): + if not isinstance(case.get(field), str) or not re.fullmatch( + r"[0-9a-f]{40}", case[field] + ): + raise ValueError(f"Case {case['id']} requires a full {field}.") + _text(case.get("skill_path"), "Skill path", 1000) + return sorted(cases, key=lambda case: case["id"]) + + +def _atomic_file(path, data, exclusive=False): + """Publish a fully written private file, never an incomplete submission.""" + fd, name = tempfile.mkstemp(prefix=".quiz-", dir=str(path.parent)) + temporary = Path(name) + try: + with os.fdopen(fd, "wb") as stream: + stream.write(data) + stream.flush() + os.fsync(stream.fileno()) + if exclusive: + os.link(str(temporary), str(path)) + else: + os.replace(str(temporary), str(path)) + finally: + temporary.unlink(missing_ok=True) + + +def freeze_questions(session: Path, manifest: dict) -> None: + """Validate and shuffle all authored questions before any live results exist.""" + session = _private_session(session) + cases = _cases(manifest) + key_path = session / "answer-key.json" + public = session / "public" + question_path = public / "questions.json" + if key_path.exists() or question_path.exists(): + raise ValueError("Questions are already frozen; create a new session instead.") + for case in cases: + runs = session / f"case-{case['id']}" / "state" / "runs" + if runs.exists() and any(runs.iterdir()): + raise ValueError("Questions must be frozen before any trial results exist.") + rng = random.Random(manifest["seed"]) + questions = [] + for case in cases: + question = _json(session / f"case-{case['id']}" / "question.json") + if not isinstance(question, dict): + raise ValueError("Question must be an object.") + stem = _text(question.get("stem"), "Question stem", 1000) + scope = _text(question.get("scope"), "Question scope", 2000) + options = question.get("options") + if not isinstance(options, list) or len(options) != 4: + raise ValueError("Each question requires exactly four options.") + checked = [] + for option in options: + if not isinstance(option, dict) or type(option.get("correct")) is not bool: + raise ValueError("Each correct flag must be a strict boolean.") + checked.append( + { + "statement": _text( + option.get("statement"), "Option statement", 600 + ), + "correct": option["correct"], + "rationale": _text( + option.get("rationale"), "Option rationale", 2000 + ), + } + ) + if ( + len( + {" ".join(option["statement"].split()).casefold() for option in checked} + ) + != 4 + ): + raise ValueError("Option statements must be distinct.") + if sum(option["correct"] for option in checked) != 1: + raise ValueError("Each question must have exactly one true option.") + rng.shuffle(checked) + labeled = [ + dict(option, letter=letter) for letter, option in zip(LETTERS, checked) + ] + questions.append( + { + "id": case["id"], + "stem": stem, + "scope": scope, + "correct": next( + option["letter"] for option in labeled if option["correct"] + ), + "commit": case, + "options": labeled, + } + ) + key = {"session_id": manifest["id"], "questions": questions} + visible = [ + { + "id": question["id"], + "stem": question["stem"], + "options": [ + {"letter": option["letter"], "statement": option["statement"]} + for option in question["options"] + ], + } + for question in questions + ] + public.mkdir(mode=0o700, exist_ok=True) + _atomic_file(key_path, _bytes(key), exclusive=True) + try: + _atomic_file(question_path, _bytes(visible), exclusive=True) + except Exception: + key_path.unlink() + raise + + +class _Node: + def __init__(self, tag, attrs=(), children=None): + self.tag = tag + self.attrs = dict(attrs) + self.children = [] if children is None else children + + def has_class(self, name): + return name in self.attrs.get("class", "").split() + + +VOID = { + "area", + "base", + "br", + "col", + "embed", + "hr", + "img", + "input", + "link", + "meta", + "param", + "source", + "track", + "wbr", +} + + +class _ReportParser(HTMLParser): + """Minimal tree parser; reject malformed nesting rather than guess a panel.""" + + def __init__(self): + super().__init__(convert_charrefs=True) + self.root = _Node("document") + self.stack = [self.root] + + def handle_starttag(self, tag, attrs): + if len({name for name, _ in attrs}) != len(attrs): + raise ValueError("Duplicate report HTML attributes.") + node = _Node(tag, [(name, value or "") for name, value in attrs]) + # The current saved renderer omits only this wrapper's closing div. + # Its explicit summary-boundary is the known end of the story; do not + # apply general browser-style error recovery to unknown structures. + if ( + tag == "p" + and node.has_class("summary-boundary") + and self.stack[-1].has_class("short-story-summary") + and self.stack[-1].tag == "div" + ): + self.stack.pop() + self.stack[-1].children.append(node) + if tag not in VOID: + self.stack.append(node) + + def handle_startendtag(self, tag, attrs): + self.handle_starttag(tag, attrs) + if tag not in VOID: + self.handle_endtag(tag) + + def handle_endtag(self, tag): + if len(self.stack) == 1 or self.stack[-1].tag != tag: + raise ValueError("Unknown or malformed report HTML nesting.") + self.stack.pop() + + def handle_data(self, data): + self.stack[-1].children.append(data) + + def finish(self, source): + self.feed(source) + self.close() + if len(self.stack) != 1: + raise ValueError("Unclosed report HTML element.") + return self.root + + +def _walk(node): + if isinstance(node, _Node): + yield node + for child in node.children: + yield from _walk(child) + + +def _children(node): + return [child for child in node.children if isinstance(child, _Node)] + + +def _plain(node): + return "".join( + _plain(child) if isinstance(child, _Node) else child for child in node.children + ) + + +def _single(nodes, label): + if len(nodes) != 1: + raise ValueError(f"Unknown report shape: expected one {label}.") + return nodes[0] + + +def _shape(node, expected): + children = _children(node) + if len(children) != len(expected) or any( + child.tag != tag or (css_class and not child.has_class(css_class)) + for child, (tag, css_class) in zip(children, expected) + ): + raise ValueError(f"Unknown summary structure inside {node.tag}.") + return children + + +SAFE_TAGS = { + "section", + "div", + "h2", + "h3", + "h4", + "h5", + "p", + "strong", + "em", + "b", + "i", + "ol", + "ul", + "li", + "span", + "details", + "summary", + "br", + "code", + "pre", + "svg", + "path", + "circle", + "rect", + "line", + "polyline", + "polygon", + "g", +} +SAFE_ATTRS = { + "class", + "role", + "aria-label", + "aria-hidden", + "viewbox", + "width", + "height", + "fill", + "stroke", + "stroke-width", + "stroke-linecap", + "stroke-linejoin", + "d", + "cx", + "cy", + "r", + "x", + "y", + "x1", + "x2", + "y1", + "y2", + "rx", + "ry", + "points", + "transform", + "focusable", + "opacity", +} +DROP_CLASSES = { + "evidence-links", + "summary-evidence-button", + "summary-provenance-source", +} + + +def _safe_fragment(node): + if isinstance(node, str): + return html.escape(node) + if node.tag in { + "nav", + "script", + "style", + "link", + "iframe", + "object", + "embed", + "img", + }: + return "" + if DROP_CLASSES.intersection(node.attrs.get("class", "").split()): + return "" + if node.tag == "a": + # Evidence navigation carries source/instruction references, not report claims. + return "" + if node.tag not in SAFE_TAGS: + raise ValueError(f"Unknown summary element: {node.tag}.") + if any( + re.search(r"url\s*\(|javascript:|data:", value, re.I) + for name, value in node.attrs.items() + if name in SAFE_ATTRS + ): + raise ValueError("Unsupported resource in summary attributes.") + attrs = "".join( + f' {name}="{html.escape(value, quote=True)}"' + for name, value in node.attrs.items() + if name in SAFE_ATTRS + ) + # SVG attribute names are case-sensitive in XML; preserve the renderer's viewBox. + attrs = attrs.replace(" viewbox=", " viewBox=") + body = "".join(_safe_fragment(child) for child in node.children) + return f"<{node.tag}{attrs}>" + ("" if node.tag in VOID else f"{body}") + + +def _blinded_report(source): + root = _ReportParser().finish(source) + nodes = list(_walk(root)) + panel = _single( + [node for node in nodes if node.attrs.get("id") == "panel-summary"], + "Summary panel", + ) + if panel.tag != "section" or not panel.has_class("panel"): + raise ValueError("Unknown Summary panel container.") + direct = _children(panel) + story = _single( + [node for node in direct if node.has_class("short-story-summary")], + "summary story", + ) + if ( + len(direct) != 5 + or direct[0] is not story + or direct[1].tag != "p" + or not direct[1].has_class("summary-boundary") + ): + raise ValueError("Unknown Summary panel sections.") + details = direct[2:] + if any( + node.tag != "details" or not node.has_class("summary-details") + for node in details + ): + raise ValueError("Unknown Summary disclosures.") + headings = [ + _single( + [node for node in _children(detail) if node.tag == "summary"], + "disclosure heading", + ) + for detail in details + ] + if ( + _plain(headings[0]).strip() != "Full scenario and expected behavior" + or _plain(headings[1]).strip() != "Other findings" + ): + raise ValueError("Unknown Summary disclosure order.") + if not any(node.has_class("evidence-limits") for node in _walk(details[2])): + raise ValueError("Missing evidence limits.") + story_children = _children(story) + if ( + len(story_children) != 2 + or story_children[0].tag != "h2" + or not story_children[0].has_class("summary-headline") + or story_children[1].tag != "ol" + or not story_children[1].has_class("story-steps") + ): + raise ValueError("Unknown summary story structure.") + steps = _children(story_children[1]) + if len(steps) != 3 or any( + node.tag != "li" or not node.has_class("story-step") for node in steps + ): + raise ValueError("Unknown story steps.") + if not any(node.has_class("intent-heading") for node in _walk(steps[0])): + raise ValueError("Unknown intended-change step.") + bodies = [ + _shape(step, [("span", "story-number"), ("div", "story-body")])[1] + for step in steps + ] + evidence_shape = [("h3", None)] + if any(node.has_class("summary-context") for node in _children(bodies[1])): + evidence_shape.append(("p", "summary-context")) + evidence_shape.extend( + [ + ("p", "summary-provenance"), + ("p", "summary-status"), + ("div", "summary-pair"), + ("nav", "evidence-nav"), + ] + ) + evidence_body = _shape(bodies[1], evidence_shape) + if _plain(evidence_body[0]) != "What the evidence shows": + raise ValueError("Unknown evidence heading.") + pair = _single( + [node for node in evidence_body if node.has_class("summary-pair")], + "summary pair", + ) + cards = _shape( + pair, + [ + ("section", "summary-before"), + ("span", "summary-arrow"), + ("section", "summary-after"), + ], + ) + for card in (cards[0], cards[2]): + children = _children(card) + empty = bool(children and children[-1].has_class("summary-empty")) + _shape( + card, + [ + ("h4", "summary-side-label"), + ("svg", "summary-picture"), + ("p", "summary-empty") if empty else ("ul", "summary-choices"), + ], + ) + meaning_shape = [("h3", None)] + for css_class in ("summary-why", "summary-caution"): + if any(node.has_class(css_class) for node in _children(bodies[2])): + meaning_shape.append(("div", css_class)) + meaning_shape.append(("ul", "summary-notices")) + meaning = _shape(bodies[2], meaning_shape) + if _plain(meaning[0]) != "What this means": + raise ValueError("Unknown meaning heading.") + for claim in meaning[1:-1]: + _shape(claim, [("strong", None), ("p", None), ("span", "evidence-links")]) + findings_shape = [("summary", None)] + if any(node.has_class("other-findings") for node in _children(details[1])): + findings_shape.append(("ul", "other-findings")) + findings_shape.extend([("p", "note"), ("nav", "evidence-nav")]) + _shape(details[1], findings_shape) + _shape(details[2], [("summary", None), ("ul", "evidence-limits")]) + # Remove the entire intent step and full scenario disclosure. Retain the renderer's + # exact generated text for headline, evidence, cards, meaning, findings and limits. + trimmed_story = _Node( + story.tag, + story.attrs.items(), + [story_children[0], _Node("ol", story_children[1].attrs.items(), steps[1:])], + ) + fragment = "".join( + _safe_fragment(node) + for node in (trimmed_story, direct[1], details[1], details[2]) + ) + styles = [node for node in nodes if node.tag == "style"] + css = _plain(_single(styles, "inline report stylesheet")) + if re.search(r"url\s*\(|@import|expression\s*\(|' + '' + "Blinded Behavior Diff summary" + + fragment + + "" + ) + + +def _load_key(session, manifest): + key = _json(session / "answer-key.json") + cases = _cases(manifest) + if not isinstance(key, dict) or key.get("session_id") != manifest["id"]: + raise ValueError("Answer key belongs to a different session.") + questions = key.get("questions") + if not isinstance(questions, list) or len(questions) != 5: + raise ValueError("Invalid answer key.") + for question, case in zip(questions, cases): + if ( + not isinstance(question, dict) + or question.get("id") != case["id"] + or question.get("commit") != case + ): + raise ValueError("Answer key case provenance does not match the session.") + _text(question.get("stem"), "Question stem", 1000) + _text(question.get("scope"), "Question scope", 2000) + options = question.get("options") + if not isinstance(options, list) or len(options) != 4: + raise ValueError("Invalid frozen options.") + for option, letter in zip(options, LETTERS): + if ( + not isinstance(option, dict) + or option.get("letter") != letter + or type(option.get("correct")) is not bool + ): + raise ValueError("Invalid frozen option labels or flags.") + _text(option.get("statement"), "Option statement", 600) + _text(option.get("rationale"), "Option rationale", 2000) + if ( + len( + {" ".join(option["statement"].split()).casefold() for option in options} + ) + != 4 + ): + raise ValueError("Frozen option statements must be distinct.") + correct = [option["letter"] for option in options if option["correct"]] + if len(correct) != 1 or question.get("correct") != correct[0]: + raise ValueError("Invalid frozen correct answer.") + return key + + +def build_quiz(session: Path, repo_root: Path) -> None: + """Build only from real saved reports, keeping source and answer key private.""" + session = _private_session(session) + manifest = _json(session / "session.json") + key = _load_key(session, manifest) + if Path(repo_root).resolve() != Path(manifest["code"]["root"]).resolve(): + raise ValueError("Build checkout does not match session provenance.") + public = session / "public" + visible = [ + { + "id": question["id"], + "stem": question["stem"], + "options": [ + {"letter": option["letter"], "statement": option["statement"]} + for option in question["options"] + ], + } + for question in key["questions"] + ] + if _json(public / "questions.json") != visible: + raise ValueError("Public questions do not match the frozen answer key.") + files = { + name: (ASSETS / name).read_bytes() + for name in ("index.html", "quiz.js", "quiz.css") + } + reports = {} + for case in _cases(manifest): + case_id = case["id"] + runs = session / f"case-{case_id}" / "state" / "runs" + candidates = ( + sorted(path for path in runs.iterdir() if path.is_dir()) + if runs.is_dir() + else [] + ) + if len(candidates) != 1: + raise ValueError( + f"Case {case_id} requires exactly one saved run; found {len(candidates)}." + ) + run = candidates[0] + record = {} + for name in ("report.html", "report.md", "decisions.json", "report-data.json"): + path = run / name + if name == "decisions.json" and not path.exists() and not path.is_symlink(): + record[name] = { + "path": str(path.relative_to(session)), + "sha256": None, + "availability": "unavailable", + } + continue + if ( + path.is_symlink() + or not path.is_file() + or session not in path.resolve().parents + ): + raise ValueError( + f"Missing or nonlocal private report: case {case_id}/{name}." + ) + data = path.read_bytes() + if not data: + raise ValueError(f"Empty private report: case {case_id}/{name}.") + if name.endswith(".json"): + json.loads(data) + record[name] = { + "path": str(path.relative_to(session)), + "sha256": _hash(data), + } + if name == "report.html": + files[f"summary-{case_id}.html"] = _blinded_report( + data.decode("utf-8") + ).encode("utf-8") + reports[str(case_id)] = record + files["session-public.json"] = _bytes( + {"session_id": manifest["id"], "case_count": 5} + ) + # Validate every input before changing any public build output. + for name, data in files.items(): + _atomic_file(public / name, data) + hashes = {name: _hash(data) for name, data in files.items()} + hashes["questions.json"] = _hash((public / "questions.json").read_bytes()) + receipt = { + "schema_version": 1, + "session_id": manifest["id"], + "public": hashes, + "reports": reports, + "answer_key_sha256": _hash((session / "answer-key.json").read_bytes()), + "session_sha256": _hash((session / "session.json").read_bytes()), + "summary_method": "saved-html-summary-panel-v1", + } + _atomic_file(session / "quiz-build.json", _bytes(receipt)) + + +def _load_build(session): + manifest = _json(session / "session.json") + key = _load_key(session, manifest) + receipt = _json(session / "quiz-build.json") + if ( + receipt.get("schema_version") != 1 + or receipt.get("session_id") != manifest["id"] + ): + raise ValueError("Invalid quiz build identity.") + for filename, field in ( + ("answer-key.json", "answer_key_sha256"), + ("session.json", "session_sha256"), + ): + if _hash((session / filename).read_bytes()) != receipt.get(field): + raise ValueError(f"Quiz build input changed: {filename}.") + expected = { + "index.html", + "quiz.js", + "quiz.css", + "questions.json", + "session-public.json", + } | {f"summary-{number}.html" for number in range(1, 6)} + if set(receipt.get("public", {})) != expected: + raise ValueError("Invalid public asset allowlist.") + files = {} + for name, digest in receipt["public"].items(): + path = session / "public" / name + if path.is_symlink() or session not in path.resolve().parents: + raise ValueError("Nonlocal quiz asset.") + data = path.read_bytes() + if _hash(data) != digest: + raise ValueError(f"Quiz asset changed: {name}.") + files["/" + name] = data + if set(receipt.get("reports", {})) != {str(number) for number in range(1, 6)}: + raise ValueError("Invalid report allowlist.") + reviews = {} + for case_id, reports in receipt["reports"].items(): + if set(reports) != { + "report.html", + "report.md", + "decisions.json", + "report-data.json", + }: + raise ValueError("Incomplete private report receipt.") + for name, record in reports.items(): + path = session / record["path"] + if path.is_symlink() or session not in path.resolve().parents: + raise ValueError("Nonlocal private report.") + if name == "decisions.json" and record.get("sha256") is None: + if ( + path.exists() + or path.is_symlink() + or record.get("availability") != "unavailable" + ): + raise ValueError("Unavailable extraction changed after quiz build.") + continue + data = path.read_bytes() + if _hash(data) != record["sha256"]: + raise ValueError("Private report changed after quiz build.") + if name == "report.html": + reviews["/review/" + case_id] = data + return manifest, key, files, reviews + + +def _validate_submission(payload, session_id): + if ( + not isinstance(payload, dict) + or set(payload) != {"session_id", "answers"} + or payload["session_id"] != session_id + ): + raise ValueError("Submission session identity is missing or incorrect.") + answers = payload["answers"] + if not isinstance(answers, list) or len(answers) != 5: + raise ValueError("Complete all five questions before submitting.") + checked = [] + for answer in answers: + if ( + not isinstance(answer, dict) + or not {"id", "letter", "confidence", "insufficient"} <= set(answer) + or set(answer) - {"id", "letter", "confidence", "insufficient", "note"} + ): + raise ValueError("Invalid answer fields.") + if type(answer["id"]) is not int or answer["id"] not in range(1, 6): + raise ValueError("Invalid case id.") + if ( + not isinstance(answer["letter"], str) + or answer["letter"] not in LETTERS + or len(answer["letter"]) != 1 + ): + raise ValueError("Choose exactly one option per question.") + if answer["confidence"] not in ("low", "medium", "high"): + raise ValueError("Confidence is required for each question.") + if type(answer["insufficient"]) is not bool: + raise ValueError("Insufficient-evidence flag must be a boolean.") + note = answer.get("note", "") + if not isinstance(note, str) or len(note) > 2000: + raise ValueError("Notes must be text of at most 2000 characters.") + checked.append(dict(answer, note=note)) + if sorted(answer["id"] for answer in checked) != list(range(1, 6)): + raise ValueError("Answer each case exactly once.") + return { + "session_id": session_id, + "answers": sorted(checked, key=lambda answer: answer["id"]), + } + + +def _score(key, submission): + submission = _validate_submission( + {name: submission[name] for name in ("session_id", "answers")}, + key["session_id"], + ) + answers = [] + for question, answer in zip(key["questions"], submission["answers"]): + selected = next( + option + for option in question["options"] + if option["letter"] == answer["letter"] + ) + sha = question["commit"]["sha"] + before = question["commit"]["before_sha"] + answers.append( + dict( + answer, + correct=answer["letter"] == question["correct"], + correct_letter=question["correct"], + statement=selected["statement"], + rationale=selected["rationale"], + scope=question["scope"], + sufficient_evidence=not answer["insufficient"], + options=question["options"], + commit=question["commit"], + source_url=f"https://github.com/DataRecce/recce-team/commit/{sha}", + before_source_url=f"https://github.com/DataRecce/recce-team/commit/{before}", + review_url=f"/review/{question['id']}", + ) + ) + return { + "session_id": key["session_id"], + "score": {"correct": sum(answer["correct"] for answer in answers), "total": 5}, + "answers": answers, + "interpretation": "This is one person's comprehension score for five sampled cases, not overall model accuracy or proof of effectiveness.", + } + + +def results(session: Path) -> dict: + session = _private_session(session) + _, key, _, _ = _load_build(session) + path = session / "submission.json" + if not path.is_file(): + raise ValueError("Results are available only after a complete submission.") + saved = _json(path) + result = _score(key, saved) + result["submitted_at"] = saved["submitted_at"] + return result + + +def make_server(session: Path, port: int = 0): + """Create a bound server, also usable by deterministic integration tests.""" + session = _private_session(session) + if type(port) is not int or not 0 <= port <= 65535: + raise ValueError("Port must be an integer from 0 to 65535.") + manifest, key, files, reviews = _load_build(session) + submission_path = session / "submission.json" + + class Handler(BaseHTTPRequestHandler): + def log_message(self, format, *args): + pass + + def _reply( + self, + status, + data, + content_type="application/json; charset=utf-8", + report=False, + ): + self.send_response(status) + self.send_header("Content-Type", content_type) + self.send_header("Content-Length", str(len(data))) + self.send_header("Cache-Control", "no-store") + self.send_header("X-Content-Type-Options", "nosniff") + self.send_header("Referrer-Policy", "no-referrer") + policy = ( + "default-src 'none'; style-src 'unsafe-inline'; script-src 'unsafe-inline'; frame-ancestors 'self'" + if report + else "default-src 'none'; script-src 'self'; style-src 'self' 'unsafe-inline'; frame-src 'self'; connect-src 'self'; frame-ancestors 'self'" + ) + self.send_header( + "Content-Security-Policy", + policy + "; base-uri 'none'; form-action 'self'", + ) + self.send_header("Connection", "close") + self.end_headers() + self.wfile.write(data) + self.close_connection = True + + def _error(self, status, message): + self._reply(status, _bytes({"error": message})) + + def _security(self, submission=False): + authority = f"127.0.0.1:{self.server.server_port}" + if self.headers.get_all("Host", []) != [authority]: + self._error(403, "Host is not this loopback evaluation server.") + return False + origins = self.headers.get_all("Origin", []) + if (origins and origins != ["http://" + authority]) or ( + submission and not origins + ): + self._error(403, "Only same-origin requests are accepted.") + return False + if self.headers.get("Sec-Fetch-Site") not in (None, "none", "same-origin"): + self._error(403, "Cross-origin requests are denied.") + return False + return True + + def _saved_result(self): + saved = _json(submission_path) + result = _score(key, saved) + result["submitted_at"] = saved["submitted_at"] + return result + + def do_GET(self): + if not self._security(): + return + path = "/index.html" if self.path == "/" else self.path + if path == "/results": + if not submission_path.is_file(): + self._error(403, "Submit all answers before viewing results.") + else: + self._reply(200, _bytes(self._saved_result())) + elif path in reviews: + if not submission_path.is_file(): + self._error(403, "Full reports remain private until submission.") + else: + self._reply( + 200, reviews[path], "text/html; charset=utf-8", report=True + ) + elif path in files: + suffix = Path(path).suffix + content_type = { + ".html": "text/html; charset=utf-8", + ".js": "text/javascript; charset=utf-8", + ".css": "text/css; charset=utf-8", + ".json": "application/json; charset=utf-8", + }[suffix] + self._reply(200, files[path], content_type) + else: + self._error(404, "Resource is not in the public allowlist.") + + def do_POST(self): + if not self._security(submission=True): + return + if self.path != "/submit": + self._error(404, "Unknown submission endpoint.") + return + if ( + self.headers.get_all("Content-Type", []) != ["application/json"] + or self.headers.get("Transfer-Encoding") is not None + ): + self._error(415, "Use application/json without transfer encoding.") + return + lengths = self.headers.get_all("Content-Length", []) + if len(lengths) != 1 or not re.fullmatch(r"[0-9]+", lengths[0]): + self._error(411, "A valid Content-Length is required.") + return + length = int(lengths[0]) + if not 1 <= length <= MAX_BODY: + self._error(413, "Submission body size is invalid.") + return + try: + raw = self.rfile.read(length) + if len(raw) != length: + raise ValueError("Incomplete request body.") + if submission_path.is_file(): + self._reply(200, _bytes(self._saved_result())) + return + payload = _validate_submission(json.loads(raw), manifest["id"]) + payload["submitted_at"] = datetime.now(timezone.utc).isoformat() + try: + _atomic_file(submission_path, _bytes(payload), exclusive=True) + except FileExistsError: + pass # Another complete first submission won the atomic publication. + self._reply(200, _bytes(self._saved_result())) + except (ValueError, KeyError, UnicodeError) as exc: + self._error(400, str(exc)) + + def do_OPTIONS(self): + if self._security(): + self._error(405, "Cross-origin access is not supported.") + + server = ThreadingHTTPServer(("127.0.0.1", port), Handler) + server.daemon_threads = True + # A stalled body must not occupy a request thread indefinitely. + original_get_request = server.get_request + + def get_request(): + connection, address = original_get_request() + connection.settimeout(10) + return connection, address + + server.get_request = get_request + return server + + +def serve(session: Path, port: int) -> None: + with make_server(session, port) as server: + print(f"http://127.0.0.1:{server.server_port}/", flush=True) + try: + server.serve_forever() + except KeyboardInterrupt: + pass diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 609ae34..046b6e4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -78,6 +78,7 @@ jobs: run: | python3 -m py_compile \ .agents/skills/release-behavior-diff/scripts/bump-version.py \ + .agents/skills/run-behavior-diff-human-evaluation/scripts/*.py \ plugin/skills/behavior-diff/scripts/codex_model.py \ plugin/skills/behavior-diff/scripts/decisions.py \ plugin/skills/behavior-diff/scripts/render.py \ @@ -97,3 +98,8 @@ jobs: - name: Run release skill unit check run: python3 .agents/skills/release-behavior-diff/scripts/bump-version.py --check + + - name: Run human evaluation helper checks + run: | + python3 tests/human-evaluation-workflow-test.py + python3 tests/human-evaluation-quiz-test.py diff --git a/AGENTS.md b/AGENTS.md index f4746aa..a568ee7 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -61,6 +61,9 @@ when the flow changes. - To run an end-to-end demo, use the local `run-behavior-diff-demo-journey` skill. - To release Behavior Diff, use the local `release-behavior-diff` skill. +- To evaluate summary comprehension with a blind human quiz, use the local + `run-behavior-diff-human-evaluation` skill. It always samples skill changes + from `DataRecce/recce-team`; live trials require fresh model-cost approval. ## Verification @@ -80,6 +83,8 @@ bash tests/hooks-test.sh python3 plugin/skills/behavior-diff/scripts/decisions.py --check bash tests/live-report-contract.sh bash tests/release-workflow-test.sh +python3 tests/human-evaluation-workflow-test.py +python3 tests/human-evaluation-quiz-test.py ``` For Markdown-only changes, also run `git diff --check`. Do not replace these diff --git a/CODING_GUIDELINES.md b/CODING_GUIDELINES.md index f5c12e3..1135e07 100644 --- a/CODING_GUIDELINES.md +++ b/CODING_GUIDELINES.md @@ -163,6 +163,8 @@ python3 plugin/skills/behavior-diff/scripts/decisions.py --check bash tests/live-report-contract.sh bash tests/release-workflow-test.sh python3 .agents/skills/release-behavior-diff/scripts/bump-version.py --check +python3 tests/human-evaluation-workflow-test.py +python3 tests/human-evaluation-quiz-test.py git diff --check ``` diff --git a/README.md b/README.md index 0420450..ccaf029 100644 --- a/README.md +++ b/README.md @@ -374,6 +374,37 @@ For live agent journeys, see [`e2e/README.md`](e2e/README.md). Each command builds fresh reports outside the repository. After changing the report code or fixtures, run the command again to see the change. +## Manually evaluate summary quality + +From this checkout, ask your coding agent: + +> Run a Behavior Diff human evaluation. + +The local +[`run-behavior-diff-human-evaluation`](.agents/skills/run-behavior-diff-human-evaluation/SKILL.md) +skill prepares five blind, four-option questions using randomly sampled skill +changes from **`DataRecce/recce-team`**. Every new evaluation pins the current +upstream `main` and records a fresh random seed. It uses this checkout's current +Behavior Diff code, including uncommitted changes—not the installed plugin. + +The skill requires private-repository access through authenticated `gh` and fresh +approval for 30 Claude Code trials plus extraction. Trial execution is a +read-only local replay; it does not post to GitHub or change Linear issues. +Both Claude Code and Codex maintainer sessions can invoke the workflow; its +trial stack is explicitly Claude Code. + +The localhost quiz hides instruction diffs, stated intent, and commit metadata +until submission. It preserves generated summary wording, collects confidence +and insufficient-evidence feedback, saves the first complete submission, then +reveals the correct answers and full reports. To analyze a completed session, +ask the agent to **analyze the human evaluation results**. + +All private artifacts remain under `~/.behavior-diff/human-evaluations/`, outside +the checkout. The workflow is manual only; CI runs synthetic helper tests, never +live evaluations. Five cases are diagnostic evidence, not an overall accuracy +estimate. See the skill's [protocol](.agents/skills/run-behavior-diff-human-evaluation/references/workflow.md) +for sampling eligibility, preparation, consent, and scoring. + ## Release 1. Update both plugin manifests to the same `X.Y.Z` version. diff --git a/docs/architecture.md b/docs/architecture.md index 24077f4..c4ac890 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -348,6 +348,45 @@ The page includes JavaScript for evidence links, expand/collapse controls, and printing. It opens as a local file without an application server. Styles are inline, but the page also references Google Fonts. +## Manual human evaluation + +The repository-local +[`run-behavior-diff-human-evaluation`](../.agents/skills/run-behavior-diff-human-evaluation/SKILL.md) +skill is a maintainer workflow, not part of the installable plugin or edit hooks. +It uses the existing trial/extraction/rendering pipeline without changing summary +wording: + +```text +Manual request + fresh live-cost approval + -> fixed recce-team upstream + pinned main + fresh random sample + -> five synthetic fixtures and patch-grounded four-option questions + -> frozen inputs/key + current local implementation fingerprint + -> unchanged local runner: 3 Before + 3 After per case + -> saved report HTML -> blinded Summary excerpts + -> loopback quiz -> first human submission -> score and full-report reveal +``` + +`evaluate.py` owns sampling, source/fixture validation, input freezing, guarded +live execution, and provenance. `quiz.py` projects the original saved HTML without +model calls and owns the local quiz server and scoring. Agent instructions own +scenario design, question truth, and interpretation; structural validation is +not a semantic judgment of the options. + +Each new session samples five eligible single-existing-skill-file changes from +`DataRecce/recce-team`, never a handpicked commit list or another repository. +The helper records the source tip, seed, exclusions, HEAD, and source-content +fingerprint, including uncommitted implementation changes. Questions freeze +before trials; attempts cannot be silently retried or replaced after results. +The Claude-only read-only launcher disables external tools and suppresses the +runner's automatic report opening so the quiz stays blind. + +The server binds only to loopback and serves an explicit public allowlist. +Instruction intent/diff, the full scenario, and source metadata stay private +before submission. The answer key is server-side; the first complete submission +unlocks original reports and commit links. Saved sessions remain available for +analysis after the checkout changes. A score out of five measures this reader's +answers to these questions, not product-wide accuracy or execution correctness. + ## Local state and system boundaries State lives under `${BEHAVIOR_DIFF_HOME:-~/.behavior-diff}/`: @@ -356,6 +395,8 @@ State lives under `${BEHAVIOR_DIFF_HOME:-~/.behavior-diff}/`: - `nudge/`: session state that prevents repeated hook suggestions. - `runs/`: each comparison's task, configuration, project copies, traces, completion grades, extracted decisions, and generated reports. +- `human-evaluations/`: private source snapshots, frozen scenarios/questions, + original reports, quiz projections, and human submissions for manual evaluations. The runner prepares copies without changing the source project. Separate working directories are not a universal security sandbox. Tool restrictions diff --git a/tests/human-evaluation-quiz-test.py b/tests/human-evaluation-quiz-test.py new file mode 100755 index 0000000..c86bba6 --- /dev/null +++ b/tests/human-evaluation-quiz-test.py @@ -0,0 +1,634 @@ +#!/usr/bin/env python3 +"""Deterministic private-quiz contracts, with authored synthetic product reports.""" + +import copy +from concurrent.futures import ThreadPoolExecutor +import contextlib +import http.client +import importlib.util +import json +from pathlib import Path +import shutil +import tempfile +import threading +import unittest + +from report_fixtures import build_reports + + +ROOT = Path(__file__).resolve().parents[1] +HELPER = ROOT / ".agents/skills/run-behavior-diff-human-evaluation/scripts/quiz.py" +spec = importlib.util.spec_from_file_location("human_evaluation_quiz", HELPER) +quiz = importlib.util.module_from_spec(spec) +spec.loader.exec_module(quiz) + + +def write_json(path, value): + path.write_text(json.dumps(value) + "\n", encoding="utf-8") + + +def question(): + return { + "stem": "Which statement describes the bounded synthetic comparison?", + "scope": "Only the observed synthetic review decision, not general reliability.", + "options": [ + { + "statement": "The observed verdict changes.", + "correct": True, + "rationale": "The authored comparison establishes a changed verdict.", + }, + { + "statement": "The verdict is the same in every observed trial.", + "correct": False, + "rationale": "The synthetic fixture includes a changed verdict.", + }, + { + "statement": "No review record is available.", + "correct": False, + "rationale": "Synthetic review records are retained.", + }, + { + "statement": "The review proves general reliability.", + "correct": False, + "rationale": "A bounded synthetic comparison cannot establish general reliability.", + }, + ], + } + + +def session_at(path, identity="synthetic-session", seed=29): + path.mkdir(mode=0o700) + manifest = { + "schema_version": 1, + "id": identity, + "source_repo": "DataRecce/recce-team", + "source_tip": "a" * 40, + "seed": seed, + "code": {"root": str(ROOT), "head": "b" * 40, "fingerprint": "synthetic"}, + "trials": 3, + "trial_model": "opus", + "extract_model": "sonnet", + "cases": [ + { + "id": number, + "sha": str(number) * 40, + "before_sha": "0" * 40, + "skill_path": f"skills/synthetic-{number}/SKILL.md", + } + for number in range(1, 6) + ], + } + write_json(path / "session.json", manifest) + for case in manifest["cases"]: + directory = path / f"case-{case['id']}" + directory.mkdir() + write_json(directory / "question.json", question()) + return manifest + + +class FreezeTests(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.session = self.root / "one" + self.manifest = session_at(self.session) + + def test_deterministic_shuffle_and_no_public_answers(self): + quiz.freeze_questions(self.session, self.manifest) + another = self.root / "two" + manifest = session_at(another) + quiz.freeze_questions(another, manifest) + self.assertEqual( + (self.session / "answer-key.json").read_bytes(), + (another / "answer-key.json").read_bytes(), + ) + visible = json.loads((self.session / "public/questions.json").read_text()) + key = json.loads((self.session / "answer-key.json").read_text()) + self.assertEqual(len(visible), 5) + for public, private in zip(visible, key["questions"]): + self.assertEqual(set(public), {"id", "stem", "options"}) + self.assertEqual( + [option["letter"] for option in public["options"]], list("ABCD") + ) + self.assertTrue( + all( + set(option) == {"letter", "statement"} + for option in public["options"] + ) + ) + self.assertEqual(sum(option["correct"] for option in private["options"]), 1) + self.assertEqual( + private["correct"], + next( + option["letter"] + for option in private["options"] + if option["correct"] + ), + ) + self.assertEqual( + (self.session / "answer-key.json").stat().st_mode & 0o777, 0o600 + ) + self.assertEqual( + json.loads((self.session / "case-1/question.json").read_text()), question() + ) + with self.assertRaises(ValueError): + quiz.freeze_questions(self.session, self.manifest) + + def test_all_inputs_validated_before_any_key_written(self): + variants = [] + bad = question() + bad["options"][0]["correct"] = 1 + variants.append(bad) + bad = question() + bad["options"][1]["correct"] = True + variants.append(bad) + bad = question() + bad["options"][0]["correct"] = False + variants.append(bad) + bad = question() + bad["options"][1]["statement"] = " THE observed verdict changes. " + variants.append(bad) + bad = question() + bad["options"].pop() + variants.append(bad) + bad = question() + bad["options"][0]["rationale"] = " " + variants.append(bad) + bad = question() + bad["options"][0]["statement"] = "x" * 601 + variants.append(bad) + bad = question() + del bad["scope"] + variants.append(bad) + bad = question() + bad["stem"] = "" + variants.append(bad) + for bad in variants: + with self.subTest(question=bad): + write_json(self.session / "case-5/question.json", bad) + with self.assertRaises(ValueError): + quiz.freeze_questions(self.session, self.manifest) + self.assertFalse((self.session / "answer-key.json").exists()) + self.assertFalse((self.session / "public/questions.json").exists()) + (self.session / "case-5/question.json").unlink() + with self.assertRaises(FileNotFoundError): + quiz.freeze_questions(self.session, self.manifest) + + def test_report_before_freeze_is_rejected(self): + run = self.session / "case-1/state/runs/synthetic" + run.mkdir(parents=True) + (run / "report.html").write_text("Already generated") + with self.assertRaises(ValueError): + quiz.freeze_questions(self.session, self.manifest) + (run / "report.html").unlink() + (run / "trace.json").write_text("{}") + with self.assertRaises(ValueError): + quiz.freeze_questions(self.session, self.manifest) + + def test_manifest_and_session_identity_are_required(self): + for field, value in [ + ("source_repo", "another/repo"), + ("seed", True), + ("id", ""), + ]: + manifest = copy.deepcopy(self.manifest) + manifest[field] = value + with self.subTest(field=field), self.assertRaises(ValueError): + quiz.freeze_questions(self.session, manifest) + other = self.root / "other" + manifest = session_at(other, identity="another-session") + quiz.freeze_questions(self.session, self.manifest) + quiz.freeze_questions(other, manifest) + self.assertNotEqual( + json.loads((self.session / "answer-key.json").read_text())["session_id"], + json.loads((other / "answer-key.json").read_text())["session_id"], + ) + + +class ReportAndServerTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.gallery_temp = tempfile.TemporaryDirectory() + cls.gallery = Path(cls.gallery_temp.name) + build_reports(cls.gallery) + + @classmethod + def tearDownClass(cls): + cls.gallery_temp.cleanup() + + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.session = self.root / "session" + self.manifest = session_at(self.session) + quiz.freeze_questions(self.session, self.manifest) + self.copy_reports() + + def copy_reports(self, scenario="changed-result"): + for number in range(1, 6): + run = self.session / f"case-{number}/state/runs/synthetic" + run.mkdir(parents=True, exist_ok=True) + for name in ( + "report.html", + "report.md", + "decisions.json", + "report-data.json", + ): + source = self.gallery / scenario / name + destination = run / name + if source.exists(): + shutil.copyfile(source, destination) + else: + destination.unlink(missing_ok=True) + + def build(self): + quiz.build_quiz(self.session, ROOT) + + @contextlib.contextmanager + def running(self): + self.build() + server = quiz.make_server(self.session, 0) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + yield server + finally: + server.shutdown() + server.server_close() + thread.join() + + def request(self, server, path, method="GET", payload=None, headers=None): + connection = http.client.HTTPConnection( + "127.0.0.1", server.server_port, timeout=5 + ) + body = ( + payload + if isinstance(payload, bytes) + else (None if payload is None else json.dumps(payload).encode()) + ) + sent = dict(headers or {}) + if method == "POST": + sent.setdefault("Content-Type", "application/json") + sent.setdefault("Origin", f"http://127.0.0.1:{server.server_port}") + connection.request(method, path, body=body, headers=sent) + response = connection.getresponse() + data = response.read() + result = response.status, dict(response.getheaders()), data + connection.close() + return result + + def payload(self, all_correct=True): + key = json.loads((self.session / "answer-key.json").read_text()) + return { + "session_id": self.manifest["id"], + "answers": [ + { + "id": question["id"], + "letter": question["correct"] + if all_correct + else next( + letter for letter in "ABCD" if letter != question["correct"] + ), + "confidence": ["low", "medium", "high"][question["id"] % 3], + "insufficient": question["id"] == 3, + "note": "Synthetic ambiguity note" if question["id"] == 3 else "", + } + for question in key["questions"] + ], + } + + def test_actual_synthetic_report_blinding_preserves_wording(self): + self.build() + original = (self.gallery / "changed-result/report.html").read_text() + root = quiz._ReportParser().finish(original) + nodes = list(quiz._walk(root)) + blinded = (self.session / "public/summary-1.html").read_text() + blind_root = quiz._ReportParser().finish(blinded) + text = quiz._plain(blind_root) + for name in ( + "summary-headline", + "summary-context", + "summary-provenance", + "summary-status", + "summary-before", + "summary-after", + "summary-why", + "summary-caution", + "summary-notices", + "evidence-limits", + ): + for node in nodes: + if node.has_class(name): + expected = quiz._plain(node) + # Navigation is excluded; generated supported-claim text remains unchanged. + if name in ("summary-why", "summary-caution"): + expected = "".join( + quiz._plain(child) + if isinstance(child, quiz._Node) + else child + for child in node.children + if not isinstance(child, quiz._Node) + or not child.has_class("evidence-links") + ) + self.assertIn(expected, text) + for name in ("story-intent", "scenario-context", "scenario-prompt"): + for node in nodes: + if node.has_class(name) and quiz._plain(node).strip(): + self.assertNotIn(quiz._plain(node), text) + self.assertNotIn("Full scenario and expected behavior", blinded) + self.assertNotIn('id="panel-instruction"', blinded) + self.assertNotIn("', + '

external secret', + ) + blinded = quiz._blinded_report(hostile) + for secret in ("leaked", "onclick", "external secret", "hostile.invalid"): + self.assertNotIn(secret, blinded) + for altered in ( + source.replace('id="panel-summary"', 'id="unknown-panel"'), + source.replace('class="summary-pair"', 'class="unknown-pair"'), + source.replace( + "Full scenario and expected behavior", "Unexpected disclosure" + ), + source.replace("