diff --git a/.agents/skills/operational-home-layout/SKILL.md b/.agents/skills/operational-home-layout/SKILL.md index 9723fb0ab18..616312ae4b2 100644 --- a/.agents/skills/operational-home-layout/SKILL.md +++ b/.agents/skills/operational-home-layout/SKILL.md @@ -102,6 +102,7 @@ state/ volatile runtime signals; gitignored ..open-decisions-cursor per-task byte cursor and folded open-decision set bounding the OPEN DECISIONS scan's cost to new status-log appends; written only by fm-classify-lib.sh's status_open_decisions_incremental, removed by teardown, safe to delete (forces one full re-fold) .status-presentation-cursor .status-presentation-lock fleet-wide per-task status identity/byte-offset manifest (including the separate STATUS OUTCOME BACKSTOP delivered-frontier offset) and serialization lock preventing already-presented status lines from being replayed as new; owned by fm-classify-lib.sh, with each task's row retired by teardown .runpod-lifecycle-.lock per-secondmate RunPod provider lifecycle lock; never touch + ..pr-publication.lock per-task PR registration transaction lock held by fm-pr-check.sh while it publishes poll artifacts and replaces pr=/pr_head=/nm_run_id=; fm-watch.sh defers a pre-metadata poll only while it is fresh; never touch runpod-omp-auth/ workstation OMP broker, read-only facade, and per-pod tunnel supervisor records and logs; never touch .afk durable away-mode flag; present = sub-supervisor may inject escalations (set by /afk, cleared on user return) .watch.lock .wake-queue.lock watcher singleton and queue serialization locks diff --git a/.agents/skills/ship-landing/SKILL.md b/.agents/skills/ship-landing/SKILL.md index 3c73aa72313..f1504965d91 100644 --- a/.agents/skills/ship-landing/SKILL.md +++ b/.agents/skills/ship-landing/SKILL.md @@ -10,8 +10,9 @@ metadata: # Ship landing For PR-based ship tasks, the ready signal depends on mode: `no-mistakes` reports `done: PR checks green` after CI is green, while `direct-PR` reports `done: PR ` after opening the PR. -On every PR-ready signal, immediately run `bin/fm-pr-check.sh ` before reporting the result - it records `pr=` and the forge's `pr_head=` when available in the task's meta and arms the watcher's merge poll, while lock-owning reconciliation through `bin/fm-todo-project.sh --check --reconcile` is the recovery backstop for a skipped arm. -Tell the captain the PR's full URL, always the complete `https://...` link rather than a bare `#number`, a concise outcome summary, and the no-mistakes risk level when applicable. +On every PR-ready signal, immediately run `bin/fm-pr-check.sh ` before reporting the result - it owns the PR-ready gates, metadata publication, and merge-poll arming, while lock-owning reconciliation through `bin/fm-todo-project.sh --check --reconcile` is the recovery backstop for a skipped arm. +When it refuses, relay its named reason (missing criteria, a run on another branch or PR, a head the pipeline did not validate, an unfinished run, or an unmatched ask-user decision) and steer the worker or decide the finding; never hand-edit task metadata to pass it. +Tell the captain the PR's full URL, always the complete `https://...` link rather than a bare `#number`, a concise outcome summary, and the observed CI result when applicable. A captain instruction to merge is explicit authority; `yolo` is the only standing routine merge authority. For any custom `state/.check.sh` you write yourself, keep it an ordinary single-link mode-`0700` file, print one line only when firstmate should wake, print nothing otherwise, finish before `FM_CHECK_TIMEOUT`, then bind its current bytes with `bin/fm-check-register.sh ` before the watcher may execute it. Retire a custom check only through `bin/fm-check-unregister.sh ` (or `bin/fm-teardown.sh` for a spawned task); never hand-compose an `rm` with `$STATE`/`$ID`. diff --git a/.agents/skills/validation-supervision/SKILL.md b/.agents/skills/validation-supervision/SKILL.md index fe9559ada5c..72039472eed 100644 --- a/.agents/skills/validation-supervision/SKILL.md +++ b/.agents/skills/validation-supervision/SKILL.md @@ -9,9 +9,9 @@ metadata: # Validation supervision -On a ship worker's implementation-complete `done:`, follow the evidence and validation lifecycle owned by `bin/fm-receipt-check.sh` before accepting completion, returning missing or invalid criteria to the same worker. -Follow its durably recorded path, keep uncertain classifications high, and keep `direct-PR` and `local-only` outside No-Mistakes. -For high-risk `no-mistakes` work, trigger full validation on the same worker using the harness invocation owned by `harness-adapters`. +On a ship worker's implementation-complete `done:`, run `bin/fm-receipt-check.sh ` and return missing or invalid criteria to the same worker before anything else; receipts establish only that every declared acceptance criterion was accounted for and certify nothing about review, CI, No-Mistakes completion, or merge readiness. +The delivery mode fixed at intake owns what follows: `direct-PR` and `local-only` stay outside No-Mistakes, and every `no-mistakes` task gets full validation. +For `no-mistakes` work, trigger full validation on the same worker using the harness invocation owned by `harness-adapters`. The task worker that starts a no-mistakes run drives the pipeline and owns every `no-mistakes axi run` and `no-mistakes axi respond` call through the next gate or outcome. Firstmate never invokes `no-mistakes axi respond` for a crew-owned run. Once validation starts, prefer routing new requirements to follow-up work rather than expanding the current task, unless a new requirement completely invalidates the work being validated; however, the smallest downstream changes needed to keep already accepted product or engineering behavior correct, add behavioral tests where an executable contract exists, or keep documentation accurate remain within the current task even when they touch files not named at intake, and corrections required to satisfy already accepted intent are not new requirements. @@ -26,17 +26,15 @@ Once ownership is settled, validate exactly once against that final head so no o An ask-user finding returns as `needs-decision` under the canonical key owned by `bin/fm-nm-run-lib.sh`; firstmate loads `ask-user-authority` and either decides or escalates per that skill. Send the same worker one exact decision naming the decision key, step, action, affected finding IDs, instructions where needed, and exact response command, passing `--resolve-key` so the worker's open decision record closes at answer time. Require the matching `resolved` event, forbid `--yes`, and require the worker to process every synchronous return until completion or a genuinely new escalation. -PR-ready and completion apply the bound-run decision check owned by `bin/fm-nm-run-lib.sh`, with the process-evidence limitation owned by `bin/fm-classify-lib.sh`. +Follow the PR-ready and done-acceptance gates owned by `bin/fm-pr-check.sh` and `bin/fm-crew-state.sh`, including their use of the decision check owned by `bin/fm-nm-run-lib.sh` and the process-evidence limitation owned by `bin/fm-classify-lib.sh`. When that check refuses, decide each named finding per `ask-user-authority` and record the answer through `fm-send`, using the fallback append documented in `bin/fm-nm-run-lib.sh` when no open decision record remains. Resume fleet supervision immediately after the decision lands. -For ordinary findings from any No-Mistakes tier, steer the original worker to return branch custody through the supported abort and sync sequence, fix the findings itself, and update receipts. -When a finding invalidates a receipt or acceptance claim, use the receipt checker owner to record it before returning branch custody. -After the original worker's fix, return high-risk work to full validation with the updated receipts and delta context. +For ordinary findings, steer the original worker to return branch custody through the supported abort and sync sequence, fix the findings itself, and update receipts. +When a finding shows a criterion unsatisfied, have the worker record a failure receipt for that criterion before the fix and a fresh success after it; the latest receipt per criterion decides it. +After the original worker's fix, return the work to full validation with the updated receipts and delta context. -When a validating run cannot bind because the plan postdates it or the base moved mid-run, keep the current plan and use the receipt checker's supported recovery procedure, owned by the header and help of `bin/fm-receipt-check.sh`. -Use its read-only binding verdict before steering the worker to retry binding the same run. -If the lane diverged, have the worker follow the pipeline's guarded branch-reconciliation guidance before retrying; preserve pipeline custody throughout recovery. +At PR-ready, `bin/fm-pr-check.sh` proves the run from No-Mistakes' own status (task branch, PR URL, full head SHA, passed or CI-green) and records `nm_run_id`; when it refuses a head mismatch, have the worker follow the pipeline's guarded branch-reconciliation guidance and re-report rather than reconciling heads in firstmate. Judge validation by the currently attributed run step through `bin/fm-crew-state.sh`, not by shell liveness or the last status event. Running, fixing, or CI states remain working; parked approval or fix-review states require the worker to follow the active gate help; passed or checks-passed is done; failed or cancelled is failed. diff --git a/AGENTS.md b/AGENTS.md index bfbd01cb1dc..77786d518b2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -223,7 +223,7 @@ Supervise all live work under section 8. ### Selected delivery path and merge authority The selected delivery path owns its own rigor. -Every ship mode keeps the evidence gate, while `bin/fm-receipt-check.sh` owns the binary low/high classifier and validation-path mechanics used inside `no-mistakes` mode. +Every ship mode keeps the evidence gate owned by `bin/fm-receipt-check.sh`, which establishes only that every declared acceptance criterion was accounted for; `no-mistakes` mode always runs full validation, and `bin/fm-pr-check.sh` proves the run and PR identity from No-Mistakes' own status at PR-ready. Never hold work outside no-mistakes for a manual clean verdict, stack serial manual reviews, or infer authority for one from security, architecture, or risk alone. A separate review or audit is allowed only when the captain explicitly requests that deliverable or the authorized task is a knowledge-only review; one named question remains scoped to that question. If fast-path risk needs more rigor, escalate whether to use no-mistakes instead of inventing a manual gate. @@ -245,7 +245,7 @@ After an autonomous merge, give the captain a one-line full-URL or local-main ou ### Validate -Load `validation-supervision` on a ship worker's implementation-complete `done:`, whenever a ship starts or already has an active no-mistakes validation run, including a mid-run requirement change or finding, and before deciding or answering any ask-user finding; it owns the evidence gate, run ownership, supersession, finding return, and validation-state judgment. +Load `validation-supervision` on a ship worker's implementation-complete `done:`, whenever a ship starts or already has an active no-mistakes validation run, including a mid-run requirement change or finding, and before deciding or answering any ask-user finding; it owns the evidence-gate handoff, run ownership, supersession, finding return, and validation-state judgment. Firstmate never invokes `no-mistakes axi respond` for a crew-owned run. ### PR ready, landing, and teardown @@ -389,7 +389,7 @@ Preserve durable structured identifiers, dependencies, and completion artifact l ## 11. Crewmate briefs `bin/fm-brief.sh` and its help own scaffold syntax, generated variants, status protocol, delivery-mode definitions of done, and exact safety mechanics. -Every new ship brief declares stable acceptance-criterion ids and receives an append-only `evidence.jsonl`; `bin/fm-receipt-schema.sh` owns the receipt schema, `bin/fm-receipt-check.sh` owns criterion parsing and completion checking, and scout/report behavior remains separate. +Every new ship brief declares stable acceptance-criterion ids and receives an append-only `evidence.jsonl`; `bin/fm-receipt-schema.sh` owns the receipt schema, `bin/fm-receipt-check.sh` owns criterion parsing and acceptance-evidence accounting, and scout/report behavior remains separate. Use its scaffold as the contract, then replace every `{TASK}` placeholder with a clear task description, acceptance criteria, constraints, and necessary context before dispatch or seeding. Keep additions task-specific rather than repeating lifecycle instructions, and alter generated sections only when the task genuinely differs from the standard shape. diff --git a/README.md b/README.md index ab13891b68b..76487670af5 100644 --- a/README.md +++ b/README.md @@ -152,7 +152,7 @@ The preference persists for the effective Firstmate home, and toggling it off re # Minutes later: PR ready for review, captain: https://github.com/you/xyz/pull/42 - (fix flaky login test - risk: low - CI green) + (fix flaky login test - CI green) > alright merge it ``` diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index cedd3bf17dc..ffed3ec4f32 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -49,19 +49,20 @@ # recorded task metadata cannot drift apart. # Every ship scaffold also declares stable acceptance-criterion ids in an exact # "# Acceptance criteria" section and creates the append-only evidence ledger at -# data//evidence.jsonl. bin/fm-receipt-check.sh owns the section parser, -# evidence gate, conservative binary risk plan, and validation timing. +# data//evidence.jsonl. bin/fm-receipt-check.sh owns the section parser +# and the acceptance-evidence gate; receipts certify nothing about review, CI, +# No-Mistakes completion, or merge readiness. # When the repo argument resolves to a directory whose git common dir equals # this code root's git common dir (any worktree of it counts), the ship scaffold # appends reserved criterion AC99 as the section's last line; keep it and use # AC1..AC98 for task criteria. projects/ resolves under FM_HOME; unresolved # names, non-git directories, and other repos get no extra criterion. # AC99 requires bin/fm-test-run.sh --changed green and -# FM_LINT_JOBS=1 bin/fm-lint.sh clean, recorded as an evidence line with the branch -# head before validation planning. No local full-suite run is required; broad -# regression is owned by the PR GitHub CI per .no-mistakes.yaml. +# FM_LINT_JOBS=1 bin/fm-lint.sh clean, recorded as an evidence line naming the +# branch head before the implementation-complete report. No local full-suite run +# is required; broad regression is owned by the PR GitHub CI per .no-mistakes.yaml. # For local-only, AC99 drops the CI clause and binds the branch-head evidence to -# reporting "ready in branch" instead of validation planning. +# reporting "ready in branch". # Ship briefs begin with a worktree-isolation assertion before the branch step. # --mode is refused on scout and secondmate scaffolds: a scout's deliverable is a # report rather than a merge, and a charter is not a delivery contract. @@ -146,8 +147,8 @@ Before reporting implementation complete, record at least one compact receipt fo Only \`--outcome success\` evidences a criterion; \`accepted-blocked\` requires a non-empty \`--captain-exception ""\` (the date plus the captain's own words or the board key that holds them) and accounts for the criterion without evidencing it; every other structured outcome records an unevidenced negative or inconclusive result. A task with any accepted-blocked criterion is never auto-merged; state those criteria and their exception references plainly in the PR description. Run \`$FM_ROOT/bin/fm-receipt-check.sh $task_id\` and do not append \`done:\` unless its JSON status is \`complete\`. -After the implementation is committed and evidence is complete, run \`$FM_ROOT/bin/fm-receipt-check.sh $task_id --implementation-complete\` before any validation plan or implementation-complete \`done:\` report. -Receipts are audit inputs rather than proof that every claim is trustworthy; keep summaries and results compact and point to commands or artifacts when useful. +The latest receipt per criterion decides it: when a finding or a later test shows a criterion unsatisfied, record a \`--outcome failure\` receipt for it, fix, and record a fresh \`--outcome success\`. +Receipts establish that you accounted for every declared criterion; they certify nothing about review, CI, No-Mistakes completion, or merge readiness, so keep summaries and results compact and point to commands or artifacts when useful. When a cited artifact lives inside this scratch worktree (for example \`.qa/evidence//report.json\`), copy it into \`$artifact_dir/\` (gitignored, survives teardown) before \`done:\` and cite the copied path; worktree-relative paths die with the worktree. EOF @@ -158,9 +159,9 @@ EOF Delivery contract: mode=direct-PR This task ships **direct-PR**: you raise the PR yourself, without the no-mistakes pipeline. The task is complete only when committed on your branch and every declared acceptance criterion has a receipt. -When it is implemented and committed, run \`$FM_ROOT/bin/fm-receipt-check.sh $task_id --plan\`, then push your branch and open a PR with \`gh-axi\`. -After the PR opens, append \`done: PR {url}\` to the status file and stop; Firstmate's canonical PR-ready helper records the observed completion. -The \`done:\` line must carry the PR URL - it is the delivery artifact - and completion recording and teardown refuse a \`done:\` that names none. +When it is implemented and committed, push your branch and open a PR with \`gh-axi\`. +After the PR opens, append \`done: PR {url}\` to the status file and stop; Firstmate's canonical PR-ready helper registers the PR. +The \`done:\` line must carry the PR URL - it is the delivery artifact - and PR registration and teardown refuse a \`done:\` that names none. Do NOT run /no-mistakes. The configured merge authority decides whether to merge the PR; firstmate relays the outcome. EOF ;; @@ -171,8 +172,8 @@ Delivery contract: mode=local-only This task ships **local-only**: no remote, no PR, no pipeline. The task is complete only when committed on your branch \`fm/$task_id\` and every declared acceptance criterion has a receipt. Do NOT push, do NOT open a PR, do NOT merge. Keep your branch a clean fast-forward onto the current default branch - if \`main\` has advanced, rebase onto it so the eventual merge stays a fast-forward. -When it is implemented, committed, and ready in the branch, run \`$FM_ROOT/bin/fm-receipt-check.sh $task_id --plan\`, then append \`done: ready in branch fm/$task_id\` to the status file, then run \`$FM_ROOT/bin/fm-receipt-check.sh $task_id --complete --terminal-evidence branch-ready\` and stop. -The \`done:\` line must carry the \`ready in branch\` marker - it is the delivery artifact - and completion recording and teardown refuse a \`done:\` that names none. +When it is implemented, committed, and ready in the branch, append \`done: ready in branch fm/$task_id\` to the status file and stop. +The \`done:\` line must carry the \`ready in branch\` marker - it is the delivery artifact - and teardown refuses a \`done:\` that names none. The configured merge authority approves the ready branch, then firstmate merges it into local \`main\` through the guarded fast-forward path. EOF ;; @@ -182,24 +183,23 @@ EOF Delivery contract: mode=no-mistakes The task is complete only when committed on your branch and every declared acceptance criterion has a receipt. When you believe it is complete, append \`done: {summary}\` to the status file and stop. -Firstmate will then classify validation risk; follow the receipt checker's plan output and help for the exact recorded receipts-mechanical or full No-Mistakes path. +Firstmate will then start full No-Mistakes validation on this worker. You drive no-mistakes by responding to its gates, not by implementing fixes. Follow the guidance no-mistakes itself provides for the mechanics: it loads when you invoke /no-mistakes, and \`no-mistakes axi run --help\` plus the \`help\` lines in each \`axi\` response are authoritative and version-matched to the installed binary. When starting no-mistakes, make \`--intent\` preserve all relevant content from this brief's \`# Task\` section plus every later accepted Firstmate requirement, clarification, constraint, exclusion, and supersession, carrying only each requirement's current accepted form; retain direct requirements instead of substituting a diff summary, and exclude generic operational, status, delivery, and other scaffold boilerplate unless it is task-specific. -Include the exact line \`Firstmate-Validation-Generation: \` in the No-Mistakes \`--intent\`, then immediately bind the returned run id with \`$FM_ROOT/bin/fm-receipt-check.sh $task_id --bind-run --generation \` so completion can prove that exact run, generation, path, and head. -If binding refuses because the run predates the plan or a mid-run rebase moved the base, do not replan - reconcile the branch to the run's own pushed head and retry the same \`--bind-run\`; the checker binds by content identity once run ownership is proven. -Do not hand-edit, commit, or fix findings yourself while a run is active; fix ordinary findings from any validation tier only after Firstmate directs the supported abort and branch-custody return sequence. +Do not hand-edit, commit, or fix findings yourself while a run is active; fix ordinary findings only after Firstmate directs the supported abort and branch-custody return sequence, and when a finding shows a criterion unsatisfied, record its failure receipt before the fix and a fresh success after it. +Firstmate's PR-ready helper reads No-Mistakes' own status for your run - branch, PR URL, full head SHA, and outcome - so never push foreign commits over the pipeline's head. Two firstmate-specific rules layer on top of that guidance: - ask-user findings are never yours to answer: escalate to firstmate (rule 6) and stop. - Report each parked gate as \`needs-decision [key=nm--]: ask-user findings=,,...\` naming every ask-user finding id the gate presents; the completion gate refuses any recorded ask-user resolution that lacks a matching firstmate \`resolved [key=nm--]\` record. + Report each parked gate as \`needs-decision [key=nm--]: ask-user findings=,,...\` naming every ask-user finding id the gate presents; PR registration and done acceptance refuse any recorded ask-user resolution that lacks a matching firstmate \`resolved [key=nm--]\` record. Firstmate applies \`ask-user-authority\` and obtains any required captain decision. When the decision comes back, feed it to the gate with \`no-mistakes axi respond\` and let the pipeline apply it - do not route the question to "the user" or implement the fix yourself. - Avoid \`--yes\`: it would silently bypass firstmate's authority check and any required captain escalation. -After /no-mistakes reports CI green (the CI-ready return point - do not wait for it to keep monitoring in the background until merge), append \`done: PR {url} checks green\` to the status file, then run \`$FM_ROOT/bin/fm-receipt-check.sh $task_id --complete --terminal-evidence no-mistakes-passed\`, and stop. You are finished. -The \`done:\` line must carry the PR URL - it is the delivery artifact - and completion recording and teardown refuse a \`done:\` that names none. +After /no-mistakes reports CI green (the CI-ready return point - do not wait for it to keep monitoring in the background until merge), append \`done: PR {url} checks green\` to the status file and stop. You are finished. +The \`done:\` line must carry the PR URL - it is the delivery artifact - and PR registration and teardown refuse a \`done:\` that names none. EOF ;; esac @@ -476,7 +476,7 @@ if repo_is_firstmate_code_root "$REPO"; then ;; *) # shellcheck disable=SC2016 # single quotes are deliberate: the backticks are literal brief text - FIRSTMATE_VERIFICATION_AC='- AC99: changed tests green via `bin/fm-test-run.sh --changed` and `FM_LINT_JOBS=1 bin/fm-lint.sh` clean, recorded as an evidence line with the branch head before validation planning; no local full-suite run is required; broad regression is owned by the PR GitHub CI per .no-mistakes.yaml.' + FIRSTMATE_VERIFICATION_AC='- AC99: changed tests green via `bin/fm-test-run.sh --changed` and `FM_LINT_JOBS=1 bin/fm-lint.sh` clean, recorded as an evidence line naming the branch head before the implementation-complete report; no local full-suite run is required; broad regression is owned by the PR GitHub CI per .no-mistakes.yaml.' ;; esac fi diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 025543286be..754912f1202 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -170,8 +170,8 @@ _fm_classify_require_pr_lib() { # 0 when the given done: status line carries the delivery artifact its # kind/mode contract requires, so a bare `done:` can never be accepted as a # delivery claim. The worker-facing contract is generated by bin/fm-brief.sh; -# this predicate is the single machine check behind both -# bin/fm-receipt-check.sh --complete and bin/fm-teardown.sh's refusal: +# this predicate is the single machine check behind bin/fm-teardown.sh's +# refusal: # ship, PR modes: a canonical PR/MR URL token (fm-pr-lib.sh's grammar) # ship, local-only: the "ready in branch" marker # scout: the report path data//report.md - an absolute path to diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index f6214cfd821..9ae1103ba75 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -9,10 +9,15 @@ # still does not describe the crew's current state as it resumes, fixes, or # re-validates. This helper never infers the current state from a tail of the log: # it reads the authoritative source (a -# no-mistakes run-step attributed under bin/fm-nm-run-lib.sh's contract, with a -# current-generation, path-matching completion receipt at the exact current -# head allowing pipeline advances and content-identity recovery, else the pane -# busy-signature) and reconciles the possibly-stale log against it. +# no-mistakes run-step attributed under bin/fm-nm-run-lib.sh's contract, else +# the pane busy-signature) and reconciles the possibly-stale log against it. +# A ship `done` is accepted only through the delivery gate in emit(): a clean +# worktree, complete acceptance evidence (bin/fm-receipt-check.sh), pr= recorded +# by bin/fm-pr-check.sh for the PR modes or a clean checked-out fm/ branch +# for local-only. For no-mistakes, it also applies the ask-user decision audit +# owned by bin/fm-nm-run-lib.sh when a recorded or attributed full run is +# available; absence of that identity does not itself refuse done. PR-ready +# registration owns the required audit before recording the PR. # # The determinism lives entirely here - only run-step / pane / log reads plus # fixed mapping logic, no heuristics and no LLM. Output is one stable, parseable, @@ -82,7 +87,7 @@ SEP=' ยท ' # Emit the one canonical line and exit 0. Detail is optional. emit() { # [detail] - local state=$1 source=$2 detail=${3:-} gate_detail line generation completed_generation validation_head completed_head validation_path completed_path current_head mode implementation_completed implementation_head requires_validation=0 completion_is_current=0 + local state=$1 source=$2 detail=${3:-} gate_detail line mode pr audit_run audit_rc if [ "$state" = 'done' ] && [ "${KIND:-}" = ship ]; then if ! fm_worktree_is_clean "${WT:-}"; then state=parked @@ -98,59 +103,44 @@ emit() { # [detail] } fi if [ "$state" = 'done' ] && [ "${KIND:-}" = ship ]; then - implementation_completed=$(grep '^implementation_completed_at=' "$META" | tail -1 | cut -d= -f2- || true) - implementation_head=$(grep '^implementation_completed_head=' "$META" | tail -1 | cut -d= -f2- || true) - mode=$(grep '^mode=' "$META" | tail -1 | cut -d= -f2- || true) - current_head=$(git -C "${WT:-}" rev-parse --verify 'HEAD^{commit}' 2>/dev/null || true) - generation=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) - completed_generation=$(grep '^validation_completed_generation=' "$META" | tail -1 | cut -d= -f2- || true) - validation_path=$(grep '^validation_path=' "$META" | tail -1 | cut -d= -f2- || true) - completed_path=$(grep '^validation_completed_path=' "$META" | tail -1 | cut -d= -f2- || true) - completed_head=$(grep '^validation_completed_head=' "$META" | tail -1 | cut -d= -f2- || true) - if [ "$mode" = no-mistakes ] && [ -n "$generation" ] \ - && [ "$completed_generation" = "$generation" ] \ - && [ "$validation_path" = full-no-mistakes ] && [ "$completed_path" = "$validation_path" ] \ - && [ -n "$current_head" ] && [ "$completed_head" = "$current_head" ]; then - completion_is_current=1 - fi - case "$implementation_completed" in - ''|*[!0-9]*) - state=parked - source=implementation-gate - detail='implementation completion is missing or invalid for the current head' - ;; - *) - implementation_head_ok=0 - if [ -n "$current_head" ] && [ "$implementation_head" = "$current_head" ]; then - implementation_head_ok=1 - elif [ "$completion_is_current" -eq 1 ] \ - && git -C "${WT:-}" rev-parse --verify "$implementation_head^{commit}" >/dev/null 2>&1; then - implementation_head_ok=1 + mode=$(meta_value mode) + pr=$(meta_value pr) + case "$mode" in + no-mistakes|direct-PR) + if [ -z "$pr" ]; then + state=parked + source=delivery-gate + detail='PR not registered; run fm-pr-check on the PR-ready report' fi - if [ "$implementation_head_ok" -ne 1 ]; then + ;; + local-only) + if [ "${CREW_BRANCH:-}" != "fm/$ID" ]; then state=parked - source=implementation-gate - detail='implementation completion is missing or stale for the current head' + source=delivery-gate + detail="branch fm/$ID is not checked out" fi ;; esac fi - if [ "$state" = 'done' ] && [ "${KIND:-}" = ship ]; then - [ -z "$generation" ] || requires_validation=1 - if [ "$source" = 'run-step' ] && [ "$mode" = no-mistakes ]; then requires_validation=1; fi - case "$mode" in direct-PR|local-only) requires_validation=1 ;; esac - validation_head=$(grep '^validation_head=' "$META" | tail -1 | cut -d= -f2- || true) - validation_head_ok=0 - if [ "$completed_head" = "$validation_head" ] && [ "$current_head" = "$validation_head" ]; then - validation_head_ok=1 - elif [ "$completion_is_current" -eq 1 ] && [ -n "$validation_head" ]; then - validation_head_ok=1 + # The done-time ask-user decision audit: the same predicate PR-ready applies, + # against the recorded nm_run_id or, before registration, the attributed run. + if [ "$state" = 'done' ] && [ "${KIND:-}" = ship ] && [ "$mode" = no-mistakes ]; then + audit_run=$(meta_value nm_run_id) + if [ -z "$audit_run" ] && [ "${HAVE_RUN:-0}" = 1 ] && [ "${RUN_SOURCE:-}" = full ]; then + audit_run=$(strip_quotes "$(nm_field id)") fi - if [ "$requires_validation" -eq 1 ] && { [ -z "$generation" ] || [ "$completed_generation" != "$generation" ] \ - || [ "$validation_head_ok" -ne 1 ] || [ "$completed_path" != "$validation_path" ]; }; then - state=parked - source=validation-gate - detail='validation completion is missing or stale for the current plan' + if [ -n "$audit_run" ]; then + audit_rc=0 + fm_nm_ask_user_decisions "$WT" "$NM_TIMEOUT" "$audit_run" "$LOG" >/dev/null || audit_rc=$? + if [ "$audit_rc" -eq 1 ]; then + state=parked + source=decision-gate + detail="run $audit_run resolved ask-user findings without matching firstmate decisions" + elif [ "$audit_rc" -ne 0 ]; then + state=parked + source=decision-gate + detail="run $audit_run ask-user decision evidence could not be read" + fi fi fi line="state: $state${SEP}source: $source" diff --git a/bin/fm-nm-run-lib.sh b/bin/fm-nm-run-lib.sh index 5e638c93abf..f3ff647f640 100644 --- a/bin/fm-nm-run-lib.sh +++ b/bin/fm-nm-run-lib.sh @@ -2,15 +2,16 @@ # Shared no-mistakes axi run attribution primitives. # # ONE owner for the no-mistakes run-attribution primitives used by -# fm-crew-state.sh (read-only current-state reporting), fm-teardown.sh -# (pre-teardown run abort, see its "Fix 1" header comment), -# fm-receipt-check.sh (bound-run completion), and fm-pr-check.sh (PR-ready -# decision evidence). Teardown uses only strict -# branch-and-head identity; crew-state additionally permits the active -# pipeline-owned exemption defined below, and receipt-check's active-advance -# ownership proof is fm_nm_run_branch_ownership. Getting this wrong in either -# direction is unsafe: a false negative hides a genuinely parked run, and a -# false positive lets teardown act on a run it does not own. +# fm-crew-state.sh (read-only current-state reporting and the done-time +# ask-user decision audit), fm-teardown.sh (pre-teardown run abort, see its +# "Fix 1" header comment), and fm-pr-check.sh (PR-ready run identity and the +# decision-evidence audit). Teardown uses only strict branch-and-head identity; +# crew-state additionally permits the active pipeline-owned exemption defined +# below. Getting this wrong in either direction is unsafe: a false negative +# hides a genuinely parked run, and a false positive lets teardown act on a run +# it does not own. No helper here reconstructs what the pipeline validated +# from the worker's object store: run identity is the branch, PR URL, full +# head SHA, status, and outcome that `axi status --run` reports. # # Bounded command in dir $1, timeout $2 seconds. The bounded form preserves # stdout, stderr, and exit status; the checked form discards stderr, while @@ -86,88 +87,6 @@ fm_nm_head_matches_worktree() { # git -C "$wt" merge-base --is-ancestor "$local_full" "$run_full" 2>/dev/null } -# Print the authoritative full commit identity for a run head in worktree $1. -# Git accepts abbreviated identities only after resolving them against the -# repository object database; callers must never compare the presentation form -# emitted by `axi status` directly with a full local SHA. -fm_nm_resolve_head() { # - [ -n "$2" ] || return 1 - git -C "$1" rev-parse --verify "${2}^{commit}" 2>/dev/null -} - -# 0 when $3 is a strict descendant of $2 after both identities are resolved by -# Git in worktree $1. -fm_nm_head_descends_from() { # - local wt=$1 ancestor=$2 descendant=$3 ancestor_full descendant_full - ancestor_full=$(fm_nm_resolve_head "$wt" "$ancestor") || return 1 - descendant_full=$(fm_nm_resolve_head "$wt" "$descendant") || return 1 - [ "$ancestor_full" != "$descendant_full" ] \ - && git -C "$wt" merge-base --is-ancestor "$ancestor_full" "$descendant_full" 2>/dev/null -} - -# 0 when $3 is a faithful restamp of the validated chain from $2 in worktree $1. -# The base must be an ancestor of both heads, their commit counts must match, and -# each pair of commits in base-to-head order must carry the same tree object. -fm_nm_head_is_faithful_restamp() { # - local wt=$1 base=$2 validated=$3 candidate=$4 base_full validated_full candidate_full - local validated_list candidate_list validated_trees candidate_trees commit - base_full=$(fm_nm_resolve_head "$wt" "$base") || return 1 - validated_full=$(fm_nm_resolve_head "$wt" "$validated") || return 1 - candidate_full=$(fm_nm_resolve_head "$wt" "$candidate") || return 1 - git -C "$wt" merge-base --is-ancestor "$base_full" "$validated_full" 2>/dev/null || return 1 - git -C "$wt" merge-base --is-ancestor "$base_full" "$candidate_full" 2>/dev/null || return 1 - validated_list=$(git -C "$wt" rev-list --reverse "$base_full..$validated_full") || return 1 - candidate_list=$(git -C "$wt" rev-list --reverse "$base_full..$candidate_full") || return 1 - validated_trees=$(printf '%s\n' "$validated_list" | while IFS= read -r commit; do - [ -n "$commit" ] || continue - git -C "$wt" rev-parse --verify "${commit}^{tree}" || exit 1 - done) || return 1 - candidate_trees=$(printf '%s\n' "$candidate_list" | while IFS= read -r commit; do - [ -n "$commit" ] || continue - git -C "$wt" rev-parse --verify "${commit}^{tree}" || exit 1 - done) || return 1 - [ "$validated_trees" = "$candidate_trees" ] -} - -# 0 when resolved revisions $2 and $3 name commits in worktree $1 whose root -# tree objects are identical - byte-identical content regardless of ancestry. -# A mid-run rebase onto a newer base produces a head that is neither equal, a -# same-base restamp, nor a descendant of the planned head, while the run still -# validates the exact content checked out; callers pair this content proof with -# authoritative run ownership rather than trusting tree equality alone. -fm_nm_commits_share_tree() { # - local wt=$1 rev1=$2 rev2=$3 tree1 tree2 - tree1=$(git -C "$wt" rev-parse --verify "${rev1}^{tree}" 2>/dev/null) || return 1 - tree2=$(git -C "$wt" rev-parse --verify "${rev2}^{tree}" 2>/dev/null) || return 1 - [ "$tree1" = "$tree2" ] -} - -# 0 when $4 is accounted for by the validated chain in worktree $1. -# It matches the validated head itself, a faithful restamp of the validated -# chain from $2, a strict descendant of $3, or a strict descendant of a faithful -# restamp of the validated chain. -fm_nm_head_is_accounted() { # - local wt=$1 base=$2 validated=$3 candidate=$4 - local base_full validated_full candidate_full validated_count prefix_head prefix_count - base_full=$(fm_nm_resolve_head "$wt" "$base") || return 1 - validated_full=$(fm_nm_resolve_head "$wt" "$validated") || return 1 - candidate_full=$(fm_nm_resolve_head "$wt" "$candidate") || return 1 - [ "$candidate_full" = "$validated_full" ] && return 0 - fm_nm_head_is_faithful_restamp "$wt" "$base_full" "$validated_full" "$candidate_full" && return 0 - fm_nm_head_descends_from "$wt" "$validated_full" "$candidate_full" && return 0 - # Pipeline restamps can be followed by additional owned commits; the leading - # segment must be a faithful restamp of the validated chain from the base. - validated_count=$(git -C "$wt" rev-list --count "$base_full..$validated_full" 2>/dev/null) || return 1 - [ "$validated_count" -gt 0 ] || return 1 - prefix_head=$(git -C "$wt" rev-list --first-parent --reverse "$base_full..$candidate_full" 2>/dev/null | head -n "$validated_count" | tail -1) || return 1 - [ -n "$prefix_head" ] || return 1 - prefix_count=$(git -C "$wt" rev-list --count "$base_full..$prefix_head" 2>/dev/null) || return 1 - [ "$prefix_count" -eq "$validated_count" ] || return 1 - fm_nm_head_is_faithful_restamp "$wt" "$base_full" "$validated_full" "$prefix_head" || return 1 - git -C "$wt" merge-base --is-ancestor "$prefix_head" "$candidate_full" 2>/dev/null || return 1 - [ "$prefix_head" != "$candidate_full" ] || return 1 -} - # 0 when a run's branch presentation identifies the checked-out branch. The # no-mistakes CLI renders Firstmate's slash branch names with a hyphen, so both # authoritative spellings are accepted and no other branch is normalized. @@ -231,61 +150,15 @@ fm_nm_run_is_pipeline_owned_active() { # fm_nm_run_is_active "$1" } -# Print the proven branch-ownership state for an ACTIVE run, or nothing. -# `pipeline_owned` in captured `axi status` output $3 means the pipeline still -# holds the branch, so the head it reports is run-owned evidence. When `axi -# status` omits branch_sync, `axi sync --check` supplies the same proof: state -# pipeline_owned again, or state synchronized once the pipeline pushed its head -# back and the branch converged while the run stays active only to monitor its -# PR. The converged state is accepted only on the full sync evidence: the same -# run id, submitted_head resolving to expected head $5, current_head and the -# reported local head both resolving to the run's observed head $6, relation -# equal, and safety already_synchronized. Anything missing, stale, or -# mismatched prints nothing so callers keep refusing unproven advances. -# An empty expected-submitted accepts whatever submitted head the run reports: -# the content-identity binding path uses it for runs that submitted before the -# latest plan recorded its head, where plan linkage is proven separately by -# tree equality instead of by the submitted anchor. -fm_nm_run_branch_ownership() { # - local wt=$1 timeout_secs=$2 status_out=$3 run_id=$4 submitted=$5 current=$6 - local state sync_out sync_state sync_run sync_submitted sync_current sync_local - state=$(fm_nm_branch_sync_state "$status_out") - if [ "$state" = pipeline_owned ]; then - printf 'pipeline_owned' - return 0 - fi - sync_out=$(fm_nm_run_checked "$wt" "$timeout_secs" axi sync --check) || sync_out= - [ -n "$sync_out" ] || return 1 - sync_state=$(fm_nm_branch_sync_state "$sync_out") - case "$sync_state" in pipeline_owned|synchronized) ;; *) return 1 ;; esac - sync_run=$(fm_nm_field "$sync_out" run) - [ -n "$sync_run" ] && [ "$sync_run" = "$run_id" ] || return 1 - sync_submitted=$(fm_nm_field "$sync_out" submitted_head) - sync_current=$(fm_nm_field "$sync_out" current_head) - if [ -n "$submitted" ] && [ -n "$sync_submitted" ]; then - [ "$(fm_nm_resolve_head "$wt" "$sync_submitted" || true)" = "$submitted" ] || return 1 - fi - if [ -n "$sync_current" ]; then - [ "$(fm_nm_resolve_head "$wt" "$sync_current" || true)" = "$current" ] || return 1 - fi - if [ "$sync_state" = synchronized ]; then - [ "$(fm_nm_field "$sync_out" relation)" = equal ] || return 1 - [ "$(fm_nm_field "$sync_out" safety)" = already_synchronized ] || return 1 - [ -n "$sync_submitted" ] && [ -n "$sync_current" ] || return 1 - sync_local=$(fm_nm_resolve_head "$wt" "$(fm_nm_field "$sync_out" head)" || true) - [ -n "$sync_local" ] && [ "$sync_local" = "$current" ] || return 1 - fi - printf '%s' "$sync_state" -} - # 0 when captured `axi status` shows a run that reached a terminal PASSED state. # A terminal run has released the branch, so branch_sync no longer reports # pipeline_owned and fm_nm_run_is_pipeline_owned_active above correctly rejects -# it. Its OWN reported head is then the authority for the commits that run -# produced, including the review and doc commits its pipeline landed after the -# validated head. Callers must therefore still require that reported head to be -# the current worktree head: that is exactly what refuses foreign commits landed -# after the run finished, which the run never reports as its head. +# it. Its OWN reported head_sha is then the authority for the commits that run +# produced, including the review and doc commits its pipeline landed; the +# PR-ready owner compares that head with the forge's live PR head, which is +# exactly what refuses foreign commits pushed after the run finished. +# passed-with-override is a terminal pass carrying a Firstmate-approved test +# exception; it does not by itself mean every forge check is green. fm_nm_run_is_terminal_passed() { # local status outcome if fm_nm_run_is_active "$1"; then return 1; fi @@ -317,6 +190,18 @@ fm_nm_ci_checks_state() { # esac } +# 0 when captured `axi status` output $3 for run $4 is PR-ready: a terminal +# passed run, or an active run whose ci step has turned green per the CI log. +# PR-ready is the handoff point for landing, not proof that the run has +# terminated; merge-time green belongs to the merge owner. +fm_nm_run_is_pr_ready() { # + local status + fm_nm_run_is_terminal_passed "$3" && return 0 + status=$(fm_nm_field "$3" status) + case "$status" in ci|running) ;; *) return 1 ;; esac + [ "$(fm_nm_ci_checks_state "$1" "$2" "$4")" = green ] +} + # The canonical status-ledger key for a parked no-mistakes ask-user gate: # nm--. A worker escalates such a gate as # `needs-decision [key=nm--]` (the generated ship brief owns that @@ -325,7 +210,8 @@ fm_nm_ci_checks_state() { # # no open record left to close, firstmate appends # `resolved [key=nm--]: answered: ` itself. # fm_nm_ask_user_decisions below compares the run's recorded gate resolutions -# against those resolved records at PR-ready and completion time. +# against those resolved records at PR-ready (bin/fm-pr-check.sh) and at +# done acceptance (bin/fm-crew-state.sh). fm_nm_ask_user_key() { # printf 'nm-%s-%s' "$1" "$2" } diff --git a/bin/fm-pr-check.sh b/bin/fm-pr-check.sh index c3b1b93b4b1..3648a5a9212 100755 --- a/bin/fm-pr-check.sh +++ b/bin/fm-pr-check.sh @@ -1,14 +1,31 @@ #!/usr/bin/env bash # Record a PR-ready task: store one validated canonical pr= and the forge's -# exact pr_head= when available, atomically arm a static merge poll, then -# record PR-path validation completion only after publication succeeds. +# exact pr_head= when available, then atomically arm a static merge poll. # The watcher check source is byte-for-byte bin/fm-pr-poll.sh; task and PR data # live only in a private sidecar and are never interpolated into shell source. # A GitHub pull request URL and a GitLab merge request URL are both accepted, # including a merge request on a self-hosted GitLab instance. -# Full-no-mistakes PR-ready requires a bound validation_run_id and passes the -# decision-evidence check owned by bin/fm-nm-run-lib.sh before arming the poll; -# unreadable run data or insufficient decision records refuse registration. +# Initial ship PR registration requires complete acceptance evidence +# (bin/fm-receipt-check.sh exits 0); missing or invalid receipts +# refuse registration naming the criteria. +# A no-mistakes task additionally proves its run from No-Mistakes' own status: +# `axi status` in the task worktree must report a run whose branch is the task +# branch, whose pr is this URL, and whose full head_sha equals the forge's PR +# head, and the run must be passed or CI-green (fm_nm_run_is_pr_ready). That +# run id is recorded as nm_run_id=; the decision-evidence audit owned by +# bin/fm-nm-run-lib.sh then runs against it as that guarantee's single +# PR-ready owner, and unreadable run data or insufficient decision records +# refuse registration. Nothing here reconstructs what the pipeline validated +# from the worker's object store. +# Re-registering the recorded pr= URL refreshes pr_head= and re-arms the poll +# without re-running the handoff gates, provided a no-mistakes ship also has +# nm_run_id= recorded. Older no-mistakes records without that run identity and +# registrations of a different URL are gated in full. fm-pr-merge.sh calls this +# before every merge, and reconciliation re-arms a skipped poll. +# Publication is serialized per task through state/..pr-publication.lock +# (a mkdir lock) so a concurrent registration cannot interleave its metadata +# replacement with this one; bin/fm-watch.sh defers a pre-metadata poll while +# that lock is fresh. # Usage: fm-pr-check.sh set -eu @@ -27,7 +44,7 @@ STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" # shellcheck source=bin/fm-nm-run-lib.sh . "$SCRIPT_DIR/fm-nm-run-lib.sh" -NM_TIMEOUT=${FM_RECEIPT_NM_TIMEOUT:-10} +NM_TIMEOUT=${FM_PR_CHECK_NM_TIMEOUT:-10} case "$NM_TIMEOUT" in ''|*[!0-9]*) NM_TIMEOUT=10 ;; esac if [ "$#" -ne 2 ]; then @@ -64,12 +81,12 @@ META_SNAPSHOT=$(mktemp "$STATE/.fm-pr-meta-snapshot.XXXXXX") || exit 1 META_RECORDS=$(mktemp "$STATE/.fm-pr-meta-records.XXXXXX") || { rm -f -- "$META_SNAPSHOT"; exit 1; } META_UPDATED=$(mktemp "$STATE/.fm-pr-meta-updated.XXXXXX") \ || { rm -f -- "$META_SNAPSHOT" "$META_RECORDS"; exit 1; } -VALIDATION_LOCK= +PUBLICATION_LOCK= pr_check_cleanup() { fm_lease_guard_release || true fm_pr_poll_cleanup rm -f -- "$META_SNAPSHOT" "$META_RECORDS" "$META_UPDATED" - [ -z "$VALIDATION_LOCK" ] || rmdir "$VALIDATION_LOCK" 2>/dev/null || true + [ -z "$PUBLICATION_LOCK" ] || rmdir "$PUBLICATION_LOCK" 2>/dev/null || true } trap pr_check_cleanup EXIT trap 'exit 1' HUP INT TERM @@ -102,27 +119,66 @@ if [ "$PROVIDER" = github ] && [ -n "$WT" ] && [ -d "$WT" ] && command -v gh >/d fi fi -# Apply the shared decision-evidence check before publishing PR-ready. -# bin/fm-nm-run-lib.sh owns the check; bin/fm-classify-lib.sh owns its -# process-evidence limitation. -VALIDATION_PATH=$(grep '^validation_path=' "$META" | tail -1 | cut -d= -f2- || true) -if [ "$VALIDATION_PATH" = full-no-mistakes ]; then - ASK_USER_RUN=$(grep '^validation_run_id=' "$META" | tail -1 | cut -d= -f2- || true) - [ -n "$ASK_USER_RUN" ] || { - echo "error: full-no-mistakes PR-ready requires a bound validation_run_id" >&2 +# Every ship PR-ready requires complete acceptance evidence; a PR already +# recorded for this task passed the handoff gates at its registration. +KIND=$(grep '^kind=' "$META" | tail -1 | cut -d= -f2- || true) +MODE=$(grep '^mode=' "$META" | tail -1 | cut -d= -f2- || true) +RECORDED_PR=$(grep '^pr=' "$META" | tail -1 | cut -d= -f2- || true) +NM_RUN_ID=$(grep '^nm_run_id=' "$META" | tail -1 | cut -d= -f2- || true) +ALREADY_REGISTERED=0 +if [ "$RECORDED_PR" = "$URL" ]; then + if [ "$KIND" = ship ] && [ "$MODE" = no-mistakes ]; then + [ -z "$NM_RUN_ID" ] || ALREADY_REGISTERED=1 + else + ALREADY_REGISTERED=1 + fi +fi +if [ "$ALREADY_REGISTERED" -eq 0 ] && [ "$KIND" = ship ]; then + EVIDENCE_RC=0 + EVIDENCE_OUT=$(FM_HOME="$FM_HOME" FM_DATA_OVERRIDE="$DATA" "$SCRIPT_DIR/fm-receipt-check.sh" "$ID" 2>&1) || EVIDENCE_RC=$? + if [ "$EVIDENCE_RC" -ne 0 ]; then + EVIDENCE_DETAIL=$(printf '%s' "$EVIDENCE_OUT" | jq -r ' + if .status == "invalid" then "invalid evidence: " + (.invalid | join("; ")) + elif .status == "missing" then "missing evidence: " + (.missing | join(", ")) + else "evidence check failed" end' 2>/dev/null || printf 'evidence check failed') + echo "error: PR-ready refused for $ID: $EVIDENCE_DETAIL" >&2 exit 1 - } - ASK_USER_DIR=$WT - [ -d "$ASK_USER_DIR" ] || ASK_USER_DIR=$FM_HOME + fi +fi + +# A no-mistakes task proves its run from No-Mistakes' own status, then passes +# the decision-evidence audit owned by bin/fm-nm-run-lib.sh (process-evidence +# limitation owned by bin/fm-classify-lib.sh). +if [ "$ALREADY_REGISTERED" -eq 0 ] && [ "$KIND" = ship ] && [ "$MODE" = no-mistakes ]; then + NM_RUN_ID= + [ -n "$WT" ] && [ -d "$WT" ] || { echo "error: no-mistakes PR-ready requires the task worktree" >&2; exit 1; } + NM_OUT=$(fm_nm_run_checked "$WT" "$NM_TIMEOUT" axi status) \ + || { echo "error: No-Mistakes status could not be observed for $ID" >&2; exit 1; } + NM_RUN_ID=$(fm_nm_field "$NM_OUT" id) + NM_BRANCH=$(fm_nm_field "$NM_OUT" branch) + NM_PR=$(fm_nm_field "$NM_OUT" pr) + NM_HEAD=$(fm_nm_field "$NM_OUT" head_sha) + case "$NM_RUN_ID" in ''|*[!A-Za-z0-9._-]*) echo "error: No-Mistakes status reports no run for $ID" >&2; exit 1 ;; esac + fm_nm_branch_matches_worktree "$WT" "$NM_BRANCH" \ + || { echo "error: No-Mistakes run $NM_RUN_ID is on branch '$NM_BRANCH', not the task branch" >&2; exit 1; } + [ "$NM_PR" = "$URL" ] \ + || { echo "error: No-Mistakes run $NM_RUN_ID opened '$NM_PR', not $URL" >&2; exit 1; } + [ -n "$PR_HEAD" ] \ + || { echo "error: the forge's PR head could not be observed, so run $NM_RUN_ID cannot be matched to $URL" >&2; exit 1; } + [ "$NM_HEAD" = "$PR_HEAD" ] \ + || { echo "error: No-Mistakes run $NM_RUN_ID validated head ${NM_HEAD:-} but the PR head is $PR_HEAD; let the pipeline reconcile the branch and re-report" >&2; exit 1; } + fm_nm_run_is_pr_ready "$WT" "$NM_TIMEOUT" "$NM_OUT" "$NM_RUN_ID" \ + || { echo "error: No-Mistakes run $NM_RUN_ID is neither passed nor CI-green" >&2; exit 1; } ASK_USER_RC=0 - ASK_USER_REPORT=$(fm_nm_ask_user_decisions "$ASK_USER_DIR" "$NM_TIMEOUT" "$ASK_USER_RUN" "$STATE/$ID.status") \ + ASK_USER_REPORT=$(fm_nm_ask_user_decisions "$WT" "$NM_TIMEOUT" "$NM_RUN_ID" "$STATE/$ID.status") \ || ASK_USER_RC=$? if [ "$ASK_USER_RC" -ne 0 ]; then if [ "$ASK_USER_RC" -eq 1 ]; then - echo "error: bound No-Mistakes run $ASK_USER_RUN resolved ask-user findings without matching firstmate decisions" >&2 + echo "error: No-Mistakes run $NM_RUN_ID resolved ask-user findings without matching firstmate decisions" >&2 printf '%s\n' "$ASK_USER_REPORT" >&2 + echo "error: firstmate must record one resolved [key=nm-$NM_RUN_ID-] line per decision event in state/$ID.status" >&2 else - echo "error: bound No-Mistakes run $ASK_USER_RUN ask-user decision evidence could not be read" >&2 + echo "error: No-Mistakes run $NM_RUN_ID ask-user decision evidence could not be read" >&2 fi exit 1 fi @@ -131,22 +187,15 @@ fi fm_pr_poll_prepare "$STATE" "$ID" "$PROVIDER" "$URL" "$HOST" "$PROJECT_PATH" "$NUMBER" "$SCRIPT_DIR/fm-pr-poll.sh" \ || { echo "error: could not prepare PR poll" >&2; exit 1; } -VALIDATION_LOCK="$STATE/.$ID.validation-plan.lock" -mkdir "$VALIDATION_LOCK" 2>/dev/null \ - || { VALIDATION_LOCK=; echo "error: validation metadata is locked" >&2; exit 1; } -EXPECTED_GENERATION=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) -VALIDATION_PATH=$(grep '^validation_path=' "$META" | tail -1 | cut -d= -f2- || true) -VALIDATION_GENERATION=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) -[ "$VALIDATION_GENERATION" = "$EXPECTED_GENERATION" ] \ - || { echo "error: validation generation changed during PR registration" >&2; exit 1; } +PUBLICATION_LOCK="$STATE/.$ID.pr-publication.lock" +mkdir "$PUBLICATION_LOCK" 2>/dev/null \ + || { PUBLICATION_LOCK=; echo "error: PR publication is locked by another registration" >&2; exit 1; } printf 'pr=%s\n' "$URL" > "$META_RECORDS" || exit 1 [ -z "$PR_HEAD" ] || printf 'pr_head=%s\n' "$PR_HEAD" >> "$META_RECORDS" || exit 1 META_REPLACE_KEYS=pr,pr_head -if [ "$VALIDATION_PATH" = direct-PR ] || [ "$VALIDATION_PATH" = receipts-mechanical ]; then - [ -n "$VALIDATION_GENERATION" ] \ - || { echo "error: PR validation generation is missing" >&2; exit 1; } - printf 'validation_pr_published_generation=%s\n' "$VALIDATION_GENERATION" >> "$META_RECORDS" || exit 1 - META_REPLACE_KEYS="$META_REPLACE_KEYS,validation_pr_published_generation" +if [ -n "$NM_RUN_ID" ]; then + printf 'nm_run_id=%s\n' "$NM_RUN_ID" >> "$META_RECORDS" || exit 1 + META_REPLACE_KEYS="$META_REPLACE_KEYS,nm_run_id" fi fm_pr_poll_publish_prepared defer-metadata || { echo "error: could not publish PR poll" >&2 @@ -166,10 +215,4 @@ fm_pr_metadata_identity_parse "$META" || { fm_pr_poll_revoke_final || true; exit || { fm_pr_poll_revoke_final || true; exit 1; } fm_pr_poll_artifacts_valid "$STATE" "$ID" "$SCRIPT_DIR/fm-pr-poll.sh" \ || { fm_pr_poll_revoke_final || true; echo "error: published PR poll is invalid" >&2; exit 1; } -if [ "$VALIDATION_PATH" = direct-PR ] || [ "$VALIDATION_PATH" = receipts-mechanical ]; then - rmdir "$VALIDATION_LOCK" || { echo "error: validation metadata lock could not be released" >&2; exit 1; } - VALIDATION_LOCK= - "$SCRIPT_DIR/fm-receipt-check.sh" "$ID" --complete --terminal-evidence pr-opened >/dev/null \ - || { echo "error: PR validation completion could not be observed" >&2; exit 1; } -fi printf 'armed: state/%s.check.sh\n' "$ID" diff --git a/bin/fm-receipt-check.sh b/bin/fm-receipt-check.sh index a6a395e7de3..a36a6f60771 100755 --- a/bin/fm-receipt-check.sh +++ b/bin/fm-receipt-check.sh @@ -1,23 +1,21 @@ #!/usr/bin/env bash -# Check a ship task's acceptance-criterion evidence and plan risk-based validation. +# Check whether a ship task's declared acceptance criteria are accounted for. # # Usage: # fm-receipt-check.sh # fm-receipt-check.sh --criterion -# fm-receipt-check.sh --implementation-complete -# fm-receipt-check.sh --mechanical-ready -# fm-receipt-check.sh --bind-run --generation -# fm-receipt-check.sh --bind-check --generation -# fm-receipt-check.sh --complete --terminal-evidence -# fm-receipt-check.sh --plan [--base ] -# fm-receipt-check.sh --invalidate-claim --invalidated-criterion # fm-receipt-check.sh --parse-criteria [--require ] # -# The default command emits one compact fm-evidence-check.v1 JSON object. -# It exits 0 when every declared criterion has at least one structurally valid -# receipt, 1 when evidence is missing, and 2 for an invalid brief or ledger. -# Tasks whose metadata positively identifies them as scouts or secondmates return -# status=not-applicable without a ledger check. +# Evidence receipts establish whether the implementing worker accounted for +# every acceptance criterion the ship brief declares. They certify nothing +# about review, test coverage, CI, No-Mistakes completion, or merge readiness; +# those belong to No-Mistakes, the forge, and the delivery owners. +# +# The default command emits one compact fm-evidence-check.v2 JSON object with +# required, evidenced, accepted_blocked, missing, and invalid. It exits 0 when +# every declared criterion is accounted for, 1 when evidence is missing, and 2 +# for an invalid brief, ledger, or task contract. Tasks whose metadata +# identifies them as scouts or secondmates return status=not-applicable. # # Acceptance criteria are owned by the exact ship-brief section: # @@ -25,118 +23,31 @@ # - AC1: First required outcome. # - AC2: Second required outcome. # -# Every listed criterion is required in v1. -# IDs must be unique AC-prefixed positive integers, and placeholder descriptions -# are invalid once completion is checked. -# Only structurally valid receipts with outcome=success evidence their criterion; -# result remains descriptive, so expected observations such as 401 stay usable. -# A receipt with outcome=accepted-blocked is valid only when it carries a -# non-empty captain_exception reference recorded verbatim by fm-receipt.sh; a -# criterion whose latest receipt is a valid accepted-blocked is accounted for -# without being evidenced, so planning, readiness, and completion can proceed. -# The evidence check always reports those criteria in a distinct -# accepted_blocked list with their exception references, never in evidenced, -# and plan, readiness, and completion output surfaces them plainly so the PR -# description can state them. Firstmate never auto-merges a task with any -# accepted-blocked criterion. -# --implementation-complete records one timestamp bound to the current clean -# implementation head, refreshes it when that head changes, and remains -# idempotent for repeated calls at the same head before --plan. -# -# --plan first requires a complete evidence check, then inspects the recorded -# worktree's base..HEAD diff with a deterministic conservative classifier. -# A supplied initial --base is accepted only when it equals the repository's -# authoritative merge boundary. -# Unreadable or unresolvable authoritative inputs are refused without a plan; -# classifiable uncertainty resolves to high. -# Risk is binary: high by default, or low only for a narrow CHANGELOG-only prose -# change with file-bound strong mechanical evidence for every changed file. -# The resolved validation_tier, validation_path, reason code, base, head, size, -# and start time are appended to state/.meta for durable inspection. -# Every completion records validation_completed_head and refuses current head -# drift unless the bound No-Mistakes run accounts for the current content in one -# of four shapes: a strict descendant of the latest validation_head, a faithful -# restamp of the validation-base-to-head chain, a strict descendant of such a -# restamp, or content identity with the run's reported head. Active descendants -# require run-owned branch evidence: current -# pipeline ownership, or the fully evidenced synchronized state once the pushed -# head converged and the run only monitors its PR, both decided by the shared -# fm_nm_run_branch_ownership predicate in bin/fm-nm-run-lib.sh. Terminal -# passed runs prove the advance through their own reported head, so a terminal -# run needs no replan or fresh run to seal its own pipeline commits. A chain the -# pipeline's rebase step restamped is proved by matching every corresponding -# commit tree, even though fresh committer stamps make the reported head neither -# validation_head nor its descendant. When a mid-run rebase onto a newer base -# or a plan recorded after the run started breaks every ancestry shape, a run -# whose reported head tree is byte-identical to the checked-out tree still -# binds and completes by content identity, but only with authoritative run -# ownership: a terminal passed run, or an active run with proven -# fm_nm_run_branch_ownership branch evidence. A run recorded as predating the -# plan binds only through that content-identity shape, and --plan refuses to -# publish an identical plan while a run is still bound to it. Foreign commits -# still refuse completion because they break ancestry, count, or pairwise tree -# identity, change the checked-out tree, or lack the required run-owned branch -# evidence. --bind-check evaluates the same binding decision read-only and -# reports one fm-validation-run-binding-check.v1 JSON object with task, status -# (bindable/refused), run, binding (ancestry/content-tree/none), reason, and head. -# Bindable exits 0; evaluated refusals and missing evidence exit 1; other -# prerequisite refusals exit 2. Prerequisite refusals use binding=none, the -# diagnostic in reason, and an empty head. -# Argument errors and a missing jq remain usage/dependency errors. -# Successful --bind-run records validation_run_binding and returns it as binding -# in fm-validation-run-binding.v1. Completion freshly evaluates content identity -# for a recorded content-tree binding or an advanced head, regardless of the -# original mechanism, so a change to synchronized ownership does not reinstate -# an incompatible submitted-head anchor. Branch-ownership evidence is owned by -# fm_nm_run_branch_ownership in bin/fm-nm-run-lib.sh. -# When --plan returns path=receipts-mechanical, append fresh successful mechanical -# evidence for every changed file with: -# -# bin/fm-receipt.sh --outcome success --file -# -# Verify those fresh receipts, then push/open the PR and report its URL: -# -# bin/fm-receipt-check.sh --mechanical-ready -# git push -u origin fm/ -# gh-axi pr create ... -# done: PR -# -# Firstmate's canonical PR-ready helper then publishes the watcher and records -# final completion with `bin/fm-pr-check.sh `. -# -# --complete requires the path-specific terminal evidence named by the generated -# instructions and records that evidence with the latest plan, path, and head; -# exact bound runs may prove current checks-green readiness through the shared CI log predicate. -# Binding and full-no-mistakes completion accept passed-with-override as a -# terminal pass carrying a Firstmate-approved test exception, alongside passed -# and checks-passed; no fresh validation run is needed to complete that pass. -# passed-with-skips lacks required evidence and is not a pass; other non-pass -# outcomes and look-alikes such as passed-with-overrides or override still refuse. -# Full-no-mistakes completion also requires the decision-evidence check owned -# by bin/fm-nm-run-lib.sh; unreadable run data or insufficient decision records -# refuse completion. bin/fm-classify-lib.sh owns the process-evidence limitation. -# --invalidate-claim appends one idempotent finding-to-criterion marker to task -# metadata after confirming that the criterion and evidence contract are current. -# Delivery mode remains authoritative: direct-PR and local-only never invoke -# No-Mistakes, while no-mistakes maps low to receipts-mechanical and high to -# full-no-mistakes, and the pinned brief and metadata mode must match exactly. -# +# Every listed criterion is required. IDs must be unique AC-prefixed positive +# integers, and placeholder descriptions are invalid. +# The latest structurally valid receipt per criterion decides it: outcome=success +# evidences the criterion, outcome=accepted-blocked (with its verbatim +# captain_exception) accounts for it without evidencing it, and every other +# outcome leaves it missing. A criterion that is invalidated by a finding is +# therefore recorded as a later failure receipt and satisfied again only by a +# fresher success. result stays descriptive, so an expected observation such +# as 401 is successful evidence when the worker records outcome=success. +# A receipt naming a criterion the brief does not declare, or any malformed +# record, makes the ledger invalid rather than silently disappearing. +# accepted_blocked is always reported as a distinct list with exception +# references, never inside evidenced; firstmate never auto-merges a task with +# any accepted-blocked criterion. +# --criterion exits 0 when the id is declared by the pinned brief, else 1. +# --parse-criteria prints "\t" per criterion, or exits 1 +# with --require when the named id is absent. +# Reads go through bin/fm-receipt-store.sh's pinned snapshot so the brief and +# ledger are read under the shared ledger lock and never through symlinks. set -eu SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" -STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" -NM_TIMEOUT=${FM_RECEIPT_NM_TIMEOUT:-10} -case "$NM_TIMEOUT" in ''|*[!0-9]*) NM_TIMEOUT=10 ;; esac - -# shellcheck source=bin/fm-nm-run-lib.sh -. "$SCRIPT_DIR/fm-nm-run-lib.sh" -# shellcheck source=bin/fm-worktree-clean-lib.sh -. "$SCRIPT_DIR/fm-worktree-clean-lib.sh" -# shellcheck source=bin/fm-classify-lib.sh -. "$SCRIPT_DIR/fm-classify-lib.sh" usage() { awk ' @@ -218,13 +129,6 @@ esac ACTION=check CRITERION_QUERY= -BASE_INPUT= -TERMINAL_EVIDENCE= -RUN_ID_INPUT= -RUN_GENERATION_INPUT= -INVALIDATION_FINDING= -INVALIDATION_CRITERION= - while [ "$#" -gt 0 ]; do option=$1 shift @@ -236,128 +140,21 @@ while [ "$#" -gt 0 ]; do CRITERION_QUERY=$1 shift ;; - --plan) - [ "$ACTION" = check ] || { echo "error: choose only one action" >&2; exit 2; } - ACTION=plan - ;; - --implementation-complete) - [ "$ACTION" = check ] || { echo "error: choose only one action" >&2; exit 2; } - ACTION=implementation-complete - ;; - --mechanical-ready) - [ "$ACTION" = check ] || { echo "error: choose only one action" >&2; exit 2; } - ACTION=mechanical-ready - ;; - --complete) - [ "$ACTION" = check ] || { echo "error: choose only one action" >&2; exit 2; } - ACTION=complete - ;; - --bind-run) - [ "$#" -gt 0 ] || { echo "error: --bind-run requires a value" >&2; exit 2; } - [ "$ACTION" = check ] || { echo "error: choose only one action" >&2; exit 2; } - ACTION=bind-run - RUN_ID_INPUT=$1 - shift - ;; - --bind-check) - [ "$#" -gt 0 ] || { echo "error: --bind-check requires a value" >&2; exit 2; } - [ "$ACTION" = check ] || { echo "error: choose only one action" >&2; exit 2; } - ACTION=bind-check - RUN_ID_INPUT=$1 - shift - ;; - --invalidate-claim) - [ "$#" -gt 0 ] || { echo "error: --invalidate-claim requires a value" >&2; exit 2; } - [ "$ACTION" = check ] || { echo "error: choose only one action" >&2; exit 2; } - ACTION=invalidate-claim - INVALIDATION_FINDING=$1 - shift - ;; - --invalidated-criterion) - [ "$#" -gt 0 ] || { echo "error: --invalidated-criterion requires a value" >&2; exit 2; } - INVALIDATION_CRITERION=$1 - shift - ;; - --generation) - [ "$#" -gt 0 ] || { echo "error: --generation requires a value" >&2; exit 2; } - RUN_GENERATION_INPUT=$1 - shift - ;; - --base|--terminal-evidence) - [ "$#" -gt 0 ] || { echo "error: $option requires a value" >&2; exit 2; } - value=$1 - shift - case "$option" in - --base) BASE_INPUT=$value ;; - --terminal-evidence) TERMINAL_EVIDENCE=$value ;; - esac - ;; *) echo "error: unknown option: $option" >&2; exit 2 ;; esac done -if [ "$ACTION" != bind-run ] && [ "$ACTION" != bind-check ] && [ -n "$RUN_GENERATION_INPUT" ]; then - echo "error: --generation requires --bind-run or --bind-check" >&2 - exit 2 -fi -if [ "$ACTION" != invalidate-claim ] && [ -n "$INVALIDATION_CRITERION" ]; then - echo "error: --invalidated-criterion requires --invalidate-claim" >&2 - exit 2 -fi - -case "$ACTION" in - check|criterion|implementation-complete|mechanical-ready|bind-run|bind-check|invalidate-claim) - [ -z "$BASE_INPUT" ] || { echo "error: --base requires --plan" >&2; exit 2; } - [ -z "$TERMINAL_EVIDENCE" ] || { echo "error: --terminal-evidence requires --complete" >&2; exit 2; } - if [ "$ACTION" = bind-run ] || [ "$ACTION" = bind-check ]; then - case "$RUN_ID_INPUT" in ''|*[!A-Za-z0-9._-]*) echo "error: invalid run id" >&2; exit 2 ;; esac - [ -n "$RUN_GENERATION_INPUT" ] || { echo "error: $ACTION requires --generation" >&2; exit 2; } - fi - if [ "$ACTION" = invalidate-claim ]; then - case "$INVALIDATION_FINDING" in F[1-9]|F[1-9][0-9]*) ;; *) echo "error: invalid finding id" >&2; exit 2 ;; esac - case "$INVALIDATION_CRITERION" in AC[1-9]|AC[1-9][0-9]*) ;; *) echo "error: invalid invalidated criterion" >&2; exit 2 ;; esac - fi - ;; - complete) - [ -z "$BASE_INPUT" ] || { echo "error: --base requires --plan" >&2; exit 2; } - [ -n "$TERMINAL_EVIDENCE" ] || { echo "error: --complete requires --terminal-evidence" >&2; exit 2; } - ;; - plan) - [ -z "$TERMINAL_EVIDENCE" ] || { echo "error: --terminal-evidence requires --complete" >&2; exit 2; } - ;; -esac - command -v jq >/dev/null 2>&1 || { echo "error: jq is required" >&2; exit 2; } +command -v perl >/dev/null 2>&1 || { echo "error: perl is required" >&2; exit 2; } -binding_check_result() { - jq -cn --arg task "$ID" --arg run "$RUN_ID_INPUT" --arg verdict "$1" \ - --arg binding "$2" --arg reason "$3" --arg head "$4" \ - '{schema:"fm-validation-run-binding-check.v1",task:$task,status:$verdict,run:$run,binding:$binding,reason:$reason,head:$head}' -} - -refuse_prerequisite() { - local reason=$1 code=${2:-2} - if [ "$ACTION" = bind-check ]; then - binding_check_result refused none "$reason" '' - else - echo "error: $reason" >&2 - fi - exit "$code" -} - -TASK_DIR="$DATA/$ID" -LEDGER_PATH="$TASK_DIR/evidence.jsonl" - -command -v perl >/dev/null 2>&1 || { refuse_prerequisite "perl is required"; } TMP_ROOT=$(mktemp -d "${TMPDIR:-/tmp}/fm-receipt-check.XXXXXX") TMP_ROOT=$(CDPATH='' cd -- "$TMP_ROOT" && pwd -P) -VALIDATION_LOCK= STORE_PID= STORE_RELEASE= STORE_RELEASE_OPEN=0 STORE_READY= +# shellcheck disable=SC2329 # Registered by the EXIT trap below. cleanup() { - [ -z "$VALIDATION_LOCK" ] || rmdir "$VALIDATION_LOCK" 2>/dev/null || true if [ -n "$STORE_PID" ]; then if kill -0 "$STORE_PID" 2>/dev/null; then if [ -s "$STORE_READY" ]; then @@ -375,10 +172,6 @@ cleanup() { } trap cleanup EXIT trap 'exit 1' HUP INT TERM -release_validation_lock() { - [ -z "$VALIDATION_LOCK" ] || rmdir "$VALIDATION_LOCK" 2>/dev/null || true - VALIDATION_LOCK= -} BRIEF="$TMP_ROOT/brief.md" LEDGER="$TMP_ROOT/evidence.jsonl" META="$TMP_ROOT/task.meta" @@ -394,131 +187,55 @@ FM_DATA_OVERRIDE="$DATA" "$SCRIPT_DIR/fm-receipt-store.sh" "$ID" hold \ STORE_PID=$! while [ ! -f "$STORE_READY" ] || [ -L "$STORE_READY" ] || [ ! -s "$STORE_READY" ]; do kill -0 "$STORE_PID" 2>/dev/null \ - || { wait "$STORE_PID" 2>/dev/null || true; STORE_PID=; refuse_prerequisite "pinned evidence snapshot failed"; } + || { wait "$STORE_PID" 2>/dev/null || true; STORE_PID=; echo "error: pinned evidence snapshot failed" >&2; exit 2; } done SNAPSHOT_RC=$(sed -n '1p' "$STORE_READY") case "$SNAPSHOT_RC" in - 0) PINNED_LEDGER_EXISTS=true ;; - 3) PINNED_LEDGER_EXISTS=false ;; - 4) PINNED_LEDGER_EXISTS=false ;; - *) refuse_prerequisite "pinned evidence snapshot failed" ;; + 0) LEDGER_EXISTS=true ;; + 3|4) LEDGER_EXISTS=false ;; + *) echo "error: pinned evidence snapshot failed" >&2; exit 2 ;; esac KIND_COUNT=$(grep -c '^kind=' "$META" 2>/dev/null || true) [ "$KIND_COUNT" -eq 1 ] \ - || { refuse_prerequisite "task metadata must contain exactly one kind"; } + || { echo "error: task metadata must contain exactly one kind" >&2; exit 2; } KIND=$(sed -n 's/^kind=//p' "$META") case "$KIND" in scout|secondmate) - if [ "$ACTION" = criterion ]; then exit 1; fi - if [ "$ACTION" != check ]; then - refuse_prerequisite "validation planning applies only to ship tasks" - fi + [ "$ACTION" != criterion ] || exit 1 jq -cn --arg task "$ID" \ - '{schema:"fm-evidence-check.v1",task:$task,kind:"non-ship",status:"not-applicable",required:[],evidenced:[],missing:[],invalid:[],accepted_blocked:[],receipt_count:0,ledger_exists:false}' + '{schema:"fm-evidence-check.v2",task:$task,kind:"non-ship",status:"not-applicable",required:[],evidenced:[],accepted_blocked:[],missing:[],invalid:[]}' exit 0 ;; ship) ;; - *) refuse_prerequisite "task metadata has an invalid kind" ;; + *) echo "error: task metadata has an invalid kind" >&2; exit 2 ;; esac -append_meta_records() { - local records updated - records=$(mktemp "$TMP_ROOT/meta-records.XXXXXX") - updated=$(mktemp "$TMP_ROOT/meta-updated.XXXXXX") - cat > "$records" - FM_DATA_OVERRIDE="$DATA" FM_STATE_OVERRIDE="$STATE" \ - "$SCRIPT_DIR/fm-receipt-store.sh" "$ID" meta-append "$META" "$records" "$updated" \ - || return 1 - mv "$updated" "$META" -} - MODE_COUNT=$(grep -c '^Delivery contract: mode=' "$BRIEF" 2>/dev/null || true) if [ "$MODE_COUNT" -eq 1 ]; then MODE=$(sed -n 's/^Delivery contract: mode=//p' "$BRIEF") case "$MODE" in no-mistakes|direct-PR|local-only) ;; - *) refuse_prerequisite "ship brief has an invalid delivery contract" ;; + *) echo "error: ship brief has an invalid delivery contract" >&2; exit 2 ;; esac elif [ "$MODE_COUNT" -eq 0 ]; then - refuse_prerequisite "ship brief has no delivery contract" + echo "error: ship brief has no delivery contract" >&2; exit 2 else - refuse_prerequisite "ship brief has multiple delivery contracts" + echo "error: ship brief has multiple delivery contracts" >&2; exit 2 fi META_MODE_COUNT=$(grep -c '^mode=' "$META" 2>/dev/null || true) [ "$META_MODE_COUNT" -eq 1 ] \ - || { refuse_prerequisite "task metadata must contain exactly one concrete delivery mode"; } + || { echo "error: task metadata must contain exactly one concrete delivery mode" >&2; exit 2; } META_MODE=$(sed -n 's/^mode=//p' "$META") case "$META_MODE" in no-mistakes|direct-PR|local-only) ;; - *) refuse_prerequisite "task metadata has no concrete delivery mode" ;; + *) echo "error: task metadata has no concrete delivery mode" >&2; exit 2 ;; esac [ "$META_MODE" = "$MODE" ] \ - || { refuse_prerequisite "task metadata delivery mode contradicts the pinned ship brief"; } -CRITERIA="$TMP_ROOT/criteria.tsv" -EVIDENCED="$TMP_ROOT/evidenced" -INVALID="$TMP_ROOT/invalid" -LATEST="$TMP_ROOT/latest.jsonl" -ACCEPTED_BLOCKED="$TMP_ROOT/accepted-blocked" -ACTIVE_INVALIDATED="$TMP_ROOT/active-invalidated" -ACTIVE_INVALIDATIONS="$TMP_ROOT/active-invalidations.tsv" -ACTIVE_REQUIREMENTS="$TMP_ROOT/active-requirements.tsv" -: > "$EVIDENCED" -: > "$INVALID" -: > "$LATEST" -: > "$ACCEPTED_BLOCKED" -: > "$ACTIVE_INVALIDATED" -: > "$ACTIVE_INVALIDATIONS" -: > "$ACTIVE_REQUIREMENTS" - -"$SCRIPT_DIR/fm-receipt-check.sh" --parse-criteria "$BRIEF" > "$CRITERIA" \ - || { [ "$ACTION" != bind-check ] || refuse_prerequisite "ship brief must contain one valid '# Acceptance criteria' section with unique AC ids and no placeholders"; exit 2; } + || { echo "error: task metadata delivery mode contradicts the pinned ship brief" >&2; exit 2; } -CURRENT_GENERATION=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) -if [ -n "$CURRENT_GENERATION" ]; then - awk -v prefix="validation_claim_invalidation=$CURRENT_GENERATION:" -v invalid="$INVALID" ' - index($0, prefix) == 1 { - value=substr($0, length(prefix) + 1) - count=split(value, fields, ":") - if (count == 4 && fields[1] ~ /^F[1-9][0-9]*$/ && fields[2] ~ /^AC[1-9][0-9]*$/ \ - && fields[3] ~ /^[0-9a-f]+$/ && (length(fields[3]) == 40 || length(fields[3]) == 64) \ - && fields[4] ~ /^[0-9]+$/ && length(fields[4]) <= 18) { - print fields[2] "\t" fields[3] "\t" fields[4] - } else { - print "active invalidation record is malformed" >> invalid - } - } - ' "$META" | sort -u > "$ACTIVE_INVALIDATIONS" -fi -if [ -s "$ACTIVE_INVALIDATIONS" ]; then - cut -f1 "$ACTIVE_INVALIDATIONS" | sort -u > "$ACTIVE_INVALIDATED" - INVALIDATION_WORKTREE=$(grep '^worktree=' "$META" | tail -1 | cut -d= -f2- || true) - CURRENT_INVALIDATION_HEAD=$(git -C "$INVALIDATION_WORKTREE" rev-parse --verify 'HEAD^{commit}' 2>/dev/null || true) - while IFS= read -r invalidated_criterion; do - INVALIDATION_READY=1 - INVALIDATION_BOUNDARY=0 - while IFS=$'\t' read -r record_criterion record_head record_boundary; do - [ "$record_criterion" = "$invalidated_criterion" ] || continue - [ "$record_boundary" -le "$INVALIDATION_BOUNDARY" ] || INVALIDATION_BOUNDARY=$record_boundary - if [ -n "$CURRENT_INVALIDATION_HEAD" ] && [ "$CURRENT_INVALIDATION_HEAD" != "$record_head" ] \ - && git -C "$INVALIDATION_WORKTREE" merge-base --is-ancestor "$record_head" "$CURRENT_INVALIDATION_HEAD" 2>/dev/null; then - set +e - git -C "$INVALIDATION_WORKTREE" diff --no-ext-diff --quiet "$record_head..$CURRENT_INVALIDATION_HEAD" - INVALIDATION_DIFF_RC=$? - set -e - case "$INVALIDATION_DIFF_RC" in - 1) ;; - 0) INVALIDATION_READY=0 ;; - *) printf 'active invalidation delta could not be inspected\n' >> "$INVALID"; INVALIDATION_READY=0 ;; - esac - else - INVALIDATION_READY=0 - fi - done < "$ACTIVE_INVALIDATIONS" - [ "$INVALIDATION_READY" -eq 1 ] \ - && printf '%s\t%s\n' "$invalidated_criterion" "$INVALIDATION_BOUNDARY" >> "$ACTIVE_REQUIREMENTS" - done < "$ACTIVE_INVALIDATED" -fi +CRITERIA="$TMP_ROOT/criteria.tsv" +"$SCRIPT_DIR/fm-receipt-check.sh" --parse-criteria "$BRIEF" > "$CRITERIA" || exit 2 if [ "$ACTION" = criterion ]; then case "$CRITERION_QUERY" in @@ -529,15 +246,17 @@ if [ "$ACTION" = criterion ]; then exit $? fi -RECEIPT_COUNT=0 -LEDGER_EXISTS=$PINNED_LEDGER_EXISTS +INVALID="$TMP_ROOT/invalid" +LATEST="$TMP_ROOT/latest.jsonl" +: > "$INVALID" +: > "$LATEST" if [ "$LEDGER_EXISTS" = true ]; then line_number=0 while IFS= read -r line || [ -n "$line" ]; do line_number=$((line_number + 1)) [ -n "$line" ] || { printf 'line %s: blank JSONL record\n' "$line_number" >> "$INVALID"; continue; } if ! printf '%s\n' "$line" | "$SCRIPT_DIR/fm-receipt-schema.sh"; then - printf 'line %s: invalid v1 receipt\n' "$line_number" >> "$INVALID" + printf 'line %s: invalid receipt\n' "$line_number" >> "$INVALID" continue fi receipt_criterion=$(printf '%s' "$line" | jq -r '.criterion') @@ -545,63 +264,27 @@ if [ "$LEDGER_EXISTS" = true ]; then printf 'line %s: undeclared criterion %s\n' "$line_number" "$receipt_criterion" >> "$INVALID" continue fi - RECEIPT_COUNT=$((RECEIPT_COUNT + 1)) printf '%s\n' "$line" \ | jq -c '{criterion,outcome,captain_exception:(.captain_exception // "")}' >> "$LATEST" - EVIDENCED_NEXT="$TMP_ROOT/evidenced.next" - if [ -s "$EVIDENCED" ]; then - grep -Fxv "$receipt_criterion" "$EVIDENCED" > "$EVIDENCED_NEXT" || true - else - : > "$EVIDENCED_NEXT" - fi - mv "$EVIDENCED_NEXT" "$EVIDENCED" - if [ "$(printf '%s' "$line" | jq -r '.outcome')" = success ]; then - if grep -Fx "$receipt_criterion" "$ACTIVE_INVALIDATED" >/dev/null 2>&1; then - receipt_head=$(printf '%s' "$line" | jq -r '.head // ""') - receipt_boundary=$(awk -F '\t' -v criterion="$receipt_criterion" '$1 == criterion { print $2 }' "$ACTIVE_REQUIREMENTS") - if [ -n "$receipt_boundary" ] && [ "$RECEIPT_COUNT" -gt "$receipt_boundary" ] \ - && [ "$receipt_head" = "$CURRENT_INVALIDATION_HEAD" ]; then - printf '%s\n' "$receipt_criterion" >> "$EVIDENCED" - fi - else - printf '%s\n' "$receipt_criterion" >> "$EVIDENCED" - fi - fi done < "$LEDGER" fi -while IFS=$'\t' read -r _criterion required_boundary; do - [ "$required_boundary" -le "$RECEIPT_COUNT" ] \ - || printf 'active invalidation boundary exceeds the evidence ledger\n' >> "$INVALID" -done < "$ACTIVE_REQUIREMENTS" REQUIRED_JSON=$(cut -f1 "$CRITERIA" | jq -Rsc 'split("\n") | map(select(length > 0))') -ACCEPTED_BLOCKED_JSON=$(jq -sc --argjson required "$REQUIRED_JSON" ' +ACCOUNTING=$(jq -sc --argjson required "$REQUIRED_JSON" ' (reduce .[] as $r ({}; .[$r.criterion] = $r)) as $latest - | [$required[] | . as $c | select($latest[$c].outcome == "accepted-blocked") - | {criterion:$c, captain_exception:$latest[$c].captain_exception}] + | { + evidenced: [$required[] | select($latest[.].outcome == "success")], + accepted_blocked: [$required[] | . as $c | select($latest[$c].outcome == "accepted-blocked") + | {criterion:$c, captain_exception:$latest[$c].captain_exception}], + missing: [$required[] | select(($latest[.].outcome // "") | test("^(success|accepted-blocked)$") | not)] + } ' "$LATEST") -printf '%s' "$ACCEPTED_BLOCKED_JSON" | jq -r '.[].criterion' > "$ACCEPTED_BLOCKED" -EVIDENCED_ORDERED="$TMP_ROOT/evidenced-ordered" -MISSING="$TMP_ROOT/missing" -: > "$EVIDENCED_ORDERED" -: > "$MISSING" -while IFS=$'\t' read -r criterion _description; do - if grep -Fx "$criterion" "$ACCEPTED_BLOCKED" >/dev/null 2>&1; then - : - elif grep -Fx "$criterion" "$EVIDENCED" >/dev/null 2>&1; then - printf '%s\n' "$criterion" >> "$EVIDENCED_ORDERED" - else - printf '%s\n' "$criterion" >> "$MISSING" - fi -done < "$CRITERIA" -EVIDENCED_JSON=$(jq -Rsc 'split("\n") | map(select(length > 0))' "$EVIDENCED_ORDERED") -MISSING_JSON=$(jq -Rsc 'split("\n") | map(select(length > 0))' "$MISSING") INVALID_JSON=$(jq -Rsc 'split("\n") | map(select(length > 0))' "$INVALID") if [ -s "$INVALID" ]; then CHECK_STATUS=invalid CHECK_RC=2 -elif [ -s "$MISSING" ]; then +elif [ "$(printf '%s' "$ACCOUNTING" | jq '.missing | length')" -gt 0 ]; then CHECK_STATUS=missing CHECK_RC=1 else @@ -609,844 +292,11 @@ else CHECK_RC=0 fi -CHECK_JSON=$(jq -cn \ +jq -cn \ --arg task "$ID" \ --arg status "$CHECK_STATUS" \ - --arg ledger "$LEDGER_PATH" \ --argjson required "$REQUIRED_JSON" \ - --argjson evidenced "$EVIDENCED_JSON" \ - --argjson missing "$MISSING_JSON" \ + --argjson accounting "$ACCOUNTING" \ --argjson invalid "$INVALID_JSON" \ - --argjson accepted_blocked "$ACCEPTED_BLOCKED_JSON" \ - --argjson receipt_count "$RECEIPT_COUNT" \ - --argjson ledger_exists "$LEDGER_EXISTS" \ - '{schema:"fm-evidence-check.v1",task:$task,kind:"ship",status:$status,required:$required,evidenced:$evidenced,missing:$missing,invalid:$invalid,accepted_blocked:$accepted_blocked,receipt_count:$receipt_count,ledger:$ledger,ledger_exists:$ledger_exists}') - -if [ "$ACTION" = check ]; then - printf '%s\n' "$CHECK_JSON" - exit "$CHECK_RC" -fi - -if [ "$ACTION" = plan ] && [ -s "$ACTIVE_INVALIDATED" ]; then - [ "$(wc -l < "$ACTIVE_REQUIREMENTS" | tr -d ' ')" -eq "$(wc -l < "$ACTIVE_INVALIDATED" | tr -d ' ')" ] \ - || { echo "error: invalidated criteria require a strict non-empty follow-up delta" >&2; exit 2; } - while IFS= read -r invalidated_criterion; do - grep -Fx "$invalidated_criterion" "$EVIDENCED" >/dev/null 2>&1 \ - || { echo "error: invalidated criterion requires fresh successful evidence after its generation boundary: $invalidated_criterion" >&2; exit 2; } - done < "$ACTIVE_INVALIDATED" -fi - -if [ "$CHECK_RC" -ne 0 ] && [ "$ACTION" != invalidate-claim ]; then - if [ "$ACTION" = bind-check ]; then - refuse_prerequisite "$(printf '%s' "$CHECK_JSON" | jq -r ' - if .status == "invalid" then "invalid evidence: " + (.invalid | join("; ")) - else "missing evidence: " + (.missing | join(", ")) end - ')" "$CHECK_RC" - fi - printf '%s\n' "$CHECK_JSON" - exit "$CHECK_RC" -fi - -if [ "$ACTION" = implementation-complete ]; then - IMPLEMENTATION_WORKTREE=$(grep '^worktree=' "$META" | tail -1 | cut -d= -f2- || true) - [ -n "$IMPLEMENTATION_WORKTREE" ] && [ -d "$IMPLEMENTATION_WORKTREE" ] \ - || { echo "error: implementation worktree is missing" >&2; exit 2; } - IMPLEMENTATION_HEAD=$(git -C "$IMPLEMENTATION_WORKTREE" rev-parse --verify 'HEAD^{commit}' 2>/dev/null) \ - || { echo "error: implementation head is unavailable" >&2; exit 2; } - fm_worktree_is_clean "$IMPLEMENTATION_WORKTREE" \ - || { echo "error: implementation worktree is dirty" >&2; exit 2; } - VALIDATION_LOCK="$STATE/.$ID.validation-plan.lock" - if ! mkdir "$VALIDATION_LOCK" 2>/dev/null; then - VALIDATION_LOCK= - echo "error: implementation completion metadata is locked" >&2 - exit 2 - fi - IMPLEMENTATION_COMPLETED=$(grep '^implementation_completed_at=' "$META" | tail -1 | cut -d= -f2- || true) - RECORDED_IMPLEMENTATION_HEAD=$(grep '^implementation_completed_head=' "$META" | tail -1 | cut -d= -f2- || true) - if [ "$RECORDED_IMPLEMENTATION_HEAD" != "$IMPLEMENTATION_HEAD" ]; then - IMPLEMENTATION_COMPLETED=$(date +%s) - case "$IMPLEMENTATION_COMPLETED" in - ''|*[!0-9]*) release_validation_lock; echo "error: implementation completion timestamp could not be recorded" >&2; exit 2 ;; - esac - printf 'implementation_completed_at=%s\nimplementation_completed_head=%s\n' \ - "$IMPLEMENTATION_COMPLETED" "$IMPLEMENTATION_HEAD" | append_meta_records \ - || { release_validation_lock; echo "error: could not record implementation completion" >&2; exit 2; } - else - case "$IMPLEMENTATION_COMPLETED" in - '') release_validation_lock; echo "error: implementation completion timestamp is missing" >&2; exit 2 ;; - *[!0-9]*) release_validation_lock; echo "error: implementation completion timestamp is invalid" >&2; exit 2 ;; - esac - fi - release_validation_lock - jq -cn --arg task "$ID" --argjson completed_at "$IMPLEMENTATION_COMPLETED" --arg completed_head "$IMPLEMENTATION_HEAD" \ - --argjson accepted_blocked "$ACCEPTED_BLOCKED_JSON" \ - '{schema:"fm-implementation-completion.v1",task:$task,status:"completed",completed_at:$completed_at,completed_head:$completed_head,accepted_blocked:$accepted_blocked}' - exit 0 -fi - -if [ "$ACTION" = invalidate-claim ]; then - cut -f1 "$CRITERIA" | grep -Fx "$INVALIDATION_CRITERION" >/dev/null 2>&1 \ - || { echo "error: invalidated criterion is not declared by the ship brief" >&2; exit 2; } - INVALIDATION_GENERATION=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) - [ "${#INVALIDATION_GENERATION}" -eq 32 ] \ - || { echo "error: claim invalidation requires a current validation generation" >&2; exit 2; } - case "$INVALIDATION_GENERATION" in - *[!0-9a-f]*) echo "error: claim invalidation requires a current validation generation" >&2; exit 2 ;; - esac - INVALIDATION_WORKTREE=$(grep '^worktree=' "$META" | tail -1 | cut -d= -f2- || true) - INVALIDATION_HEAD=$(git -C "$INVALIDATION_WORKTREE" rev-parse --verify 'HEAD^{commit}' 2>/dev/null || true) - if [ -z "$INVALIDATION_HEAD" ] || ! fm_worktree_is_clean "$INVALIDATION_WORKTREE"; then - echo "error: claim invalidation requires a clean current worktree head" >&2 - exit 2 - fi - INVALIDATION_PREFIX="validation_claim_invalidation=$INVALIDATION_GENERATION:$INVALIDATION_FINDING:$INVALIDATION_CRITERION:" - INVALIDATION_MARKER="$INVALIDATION_PREFIX$INVALIDATION_HEAD:$RECEIPT_COUNT" - VALIDATION_LOCK="$STATE/.$ID.validation-plan.lock" - if ! mkdir "$VALIDATION_LOCK" 2>/dev/null; then - VALIDATION_LOCK= - echo "error: validation metadata is locked" >&2 - exit 2 - fi - EXISTING_INVALIDATION=$(awk -v prefix="$INVALIDATION_PREFIX" 'index($0, prefix) == 1 { print; exit }' "$META") - if [ -z "$EXISTING_INVALIDATION" ]; then - printf '%s\n' "$INVALIDATION_MARKER" | append_meta_records \ - || { release_validation_lock; echo "error: could not record claim invalidation" >&2; exit 2; } - fi - release_validation_lock - RECORDED_INVALIDATION=$(awk -v prefix="$INVALIDATION_PREFIX" 'index($0, prefix) == 1 { print; exit }' "$META") - RECORDED_INVALIDATION=${RECORDED_INVALIDATION#"$INVALIDATION_PREFIX"} - RECORDED_HEAD=${RECORDED_INVALIDATION%:*} - RECORDED_BOUNDARY=${RECORDED_INVALIDATION##*:} - jq -cn --arg task "$ID" --arg generation "$INVALIDATION_GENERATION" --arg finding "$INVALIDATION_FINDING" \ - --arg criterion "$INVALIDATION_CRITERION" --arg invalidated_head "$RECORDED_HEAD" --argjson receipt_boundary "$RECORDED_BOUNDARY" \ - '{schema:"fm-claim-invalidation.v1",task:$task,status:"recorded",generation:$generation,finding:$finding,criterion:$criterion,invalidated_head:$invalidated_head,receipt_boundary:$receipt_boundary}' - exit 0 -fi - -mechanical_evidence_covers_file() { - local ledger=$1 file=$2 - jq --arg file "$file" -se ' - any(.[]; .file == $file and .outcome == "success" and (.type | test("^(test|build|lint|typecheck)$"))) - ' "$ledger" >/dev/null 2>&1 -} - -if [ "$ACTION" = bind-run ] || [ "$ACTION" = bind-check ]; then - BIND_WORKTREE=$(grep '^worktree=' "$META" | tail -1 | cut -d= -f2- || true) - BIND_PATH=$(grep '^validation_path=' "$META" | tail -1 | cut -d= -f2- || true) - BIND_BASE=$(grep '^validation_base=' "$META" | tail -1 | cut -d= -f2- || true) - BIND_HEAD=$(grep '^validation_head=' "$META" | tail -1 | cut -d= -f2- || true) - BIND_GENERATION=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) - BIND_PREPLAN_RUN=$(grep '^validation_preplan_run_id=' "$META" | tail -1 | cut -d= -f2- || true) - [ "$BIND_PATH" = full-no-mistakes ] || { refuse_prerequisite "latest plan does not use full No-Mistakes"; } - [ -n "$BIND_WORKTREE" ] && [ -d "$BIND_WORKTREE" ] || { refuse_prerequisite "validation worktree is missing"; } - BIND_HEAD=$(git -C "$BIND_WORKTREE" rev-parse --verify "$BIND_HEAD^{commit}" 2>/dev/null) \ - || { refuse_prerequisite "validated head is missing"; } - BIND_BASE=$(git -C "$BIND_WORKTREE" rev-parse --verify "$BIND_BASE^{commit}" 2>/dev/null) \ - || { refuse_prerequisite "validation base is missing"; } - fm_worktree_is_clean "$BIND_WORKTREE" \ - || { refuse_prerequisite "validation worktree is dirty"; } - BIND_OUT=$(fm_nm_run_checked "$BIND_WORKTREE" "$NM_TIMEOUT" axi status --run "$RUN_ID_INPUT") \ - || { refuse_prerequisite "No-Mistakes run could not be observed"; } - BIND_OBSERVED_ID=$(fm_nm_field "$BIND_OUT" id) - BIND_OBSERVED_HEAD=$(fm_nm_field "$BIND_OUT" head) - BIND_STATUS=$(fm_nm_field "$BIND_OUT" status) - BIND_OUTCOME=$(fm_nm_field "$BIND_OUT" outcome) - # The generation is Firstmate's own plan-side nonce, recorded authoritatively - # in this task's metadata as validation_generation. Real no-mistakes does not - # echo the agent-supplied --intent body back through `axi logs --step intent` - # (it reports only "using intent supplied by the agent"), so the run is bound - # to its plan through what no-mistakes DOES report authoritatively - the run id - # and head from `axi status --run` (checked below) plus the created-after-plan - # boundary or the owned content-identity recovery proof - cross-checked - # against the plan metadata's generation. A - # superseded plan mints a new generation and clears validation_run_*, so a run - # bound under an old generation can never satisfy completion's generation check. - [ "$RUN_GENERATION_INPUT" = "$BIND_GENERATION" ] || { refuse_prerequisite "run generation does not match the latest plan"; } - BIND_STATE_OK=0 - case "$BIND_STATUS:$BIND_OUTCOME" in - failed:*|cancelled:*|*:failed|*:cancelled) ;; - passed:*|passed-with-override:*|checks-passed:*|*:passed|*:passed-with-override|*:checks-passed) BIND_STATE_OK=1 ;; - running:*|fixing:*|ci:*|awaiting_approval:*) BIND_STATE_OK=1 ;; - esac - BIND_RUN_BRANCH=$(fm_nm_field "$BIND_OUT" branch) - BIND_BRANCH_MATCH=0 - if [ -n "$BIND_RUN_BRANCH" ] && fm_nm_branch_matches_worktree "$BIND_WORKTREE" "$BIND_RUN_BRANCH"; then - BIND_BRANCH_MATCH=1 - fi - # The run's head is the planned commit itself, a faithful restamp of the - # validated chain, or a proven pipeline-owned descendant that advanced after - # the plan was recorded (review/doc/lint fix commits). Allow descendants so - # binding does not require a fresh plan for every no-mistakes fix round. - BIND_RUN_HEAD=$(fm_nm_resolve_head "$BIND_WORKTREE" "$BIND_OBSERVED_HEAD" || true) - BIND_HEAD_ACCOUNTED=0 - BIND_CONTENT_ACCOUNTED=0 - if [ -n "$BIND_RUN_HEAD" ]; then - if [ "$BIND_RUN_HEAD" = "$BIND_HEAD" ]; then - [ "$BIND_BRANCH_MATCH" -eq 1 ] && BIND_HEAD_ACCOUNTED=1 - elif fm_nm_head_is_faithful_restamp "$BIND_WORKTREE" "$BIND_BASE" "$BIND_HEAD" "$BIND_RUN_HEAD"; then - [ "$BIND_BRANCH_MATCH" -eq 1 ] && BIND_HEAD_ACCOUNTED=1 - elif fm_nm_head_is_accounted "$BIND_WORKTREE" "$BIND_BASE" "$BIND_HEAD" "$BIND_RUN_HEAD"; then - # The head advanced after the plan; require branch identity and active or - # terminal passed ownership so an unrelated descendant cannot bind. - if [ "$BIND_BRANCH_MATCH" -eq 1 ]; then - if fm_nm_run_is_terminal_passed "$BIND_OUT"; then - BIND_HEAD_ACCOUNTED=1 - elif fm_nm_run_is_active "$BIND_OUT"; then - # Active ownership is proven by the shared predicate: pipeline_owned - # while the pipeline holds the branch, or the fully evidenced - # synchronized state once the pushed-back head converged and the run - # only monitors its PR (bin/fm-nm-run-lib.sh). - branch_sync_state=$(fm_nm_run_branch_ownership "$BIND_WORKTREE" "$NM_TIMEOUT" \ - "$BIND_OUT" "$RUN_ID_INPUT" "$BIND_HEAD" "$BIND_RUN_HEAD" || true) - if [ -n "$branch_sync_state" ]; then - BIND_HEAD_ACCOUNTED=1 - fi - fi - fi - fi - # Content-identity shape: when ancestry cannot account for the run head - a - # plan recorded after the run started, or a mid-run rebase onto a newer - # base - a run whose reported head tree is byte-identical to the checked-out - # tree still binds, but only with authoritative run ownership: a terminal - # passed run, or an active run with proven branch ownership. The submitted - # anchor stays empty because a run that predates the plan or was rebased - # mid-run may have submitted a different head; plan linkage is - # proven by tree equality instead. - if [ "$BIND_BRANCH_MATCH" -eq 1 ] \ - && fm_nm_commits_share_tree "$BIND_WORKTREE" "$BIND_RUN_HEAD" HEAD; then - if fm_nm_run_is_terminal_passed "$BIND_OUT"; then - BIND_CONTENT_ACCOUNTED=1 - elif fm_nm_run_is_active "$BIND_OUT"; then - branch_sync_state=$(fm_nm_run_branch_ownership "$BIND_WORKTREE" "$NM_TIMEOUT" \ - "$BIND_OUT" "$RUN_ID_INPUT" '' "$BIND_RUN_HEAD" || true) - [ -n "$branch_sync_state" ] && BIND_CONTENT_ACCOUNTED=1 - fi - fi - fi - # The created-after-plan boundary yields only to the content proof: a run - # recorded as predating the latest plan binds solely through the - # content-identity shape, never through ancestry alone. - BIND_PREPLAN_BLOCKED=0 - if [ "$RUN_ID_INPUT" = "$BIND_PREPLAN_RUN" ] && [ "$BIND_CONTENT_ACCOUNTED" -ne 1 ]; then - BIND_PREPLAN_BLOCKED=1 - fi - BIND_ACCOUNTED=0 - { [ "$BIND_HEAD_ACCOUNTED" -eq 1 ] || [ "$BIND_CONTENT_ACCOUNTED" -eq 1 ]; } && BIND_ACCOUNTED=1 - BIND_MECHANISM=none - if [ "$BIND_CONTENT_ACCOUNTED" -eq 1 ] \ - && { [ "$BIND_HEAD_ACCOUNTED" -eq 0 ] || [ "$RUN_ID_INPUT" = "$BIND_PREPLAN_RUN" ]; }; then - BIND_MECHANISM=content-tree - elif [ "$BIND_HEAD_ACCOUNTED" -eq 1 ]; then - BIND_MECHANISM=ancestry - elif [ "$BIND_CONTENT_ACCOUNTED" -eq 1 ]; then - BIND_MECHANISM=content-tree - fi - BINDABLE=0 - if [ "$BIND_OBSERVED_ID" = "$RUN_ID_INPUT" ] && [ "$BIND_STATE_OK" -eq 1 ] \ - && [ "$BIND_ACCOUNTED" -eq 1 ] && [ "$BIND_PREPLAN_BLOCKED" -eq 0 ]; then - BINDABLE=1 - fi - BIND_REASON= - if [ "$BINDABLE" -eq 1 ]; then - : - elif [ "$BIND_OBSERVED_ID" != "$RUN_ID_INPUT" ]; then - BIND_REASON="run-id-mismatch" - elif [ "$BIND_STATE_OK" -ne 1 ]; then - BIND_REASON="run-not-in-bindable-state" - elif [ "$BIND_PREPLAN_BLOCKED" -eq 1 ]; then - BIND_REASON="predates-latest-plan" - elif [ -z "$BIND_RUN_HEAD" ]; then - BIND_REASON="run-head-unresolvable" - elif [ "$BIND_BRANCH_MATCH" -ne 1 ]; then - BIND_REASON="run-branch-mismatch" - elif fm_nm_commits_share_tree "$BIND_WORKTREE" "$BIND_RUN_HEAD" HEAD; then - BIND_REASON="ownership-unproven" - else - BIND_REASON="head-content-not-accounted" - fi - if [ "$ACTION" = bind-check ]; then - BIND_VERDICT=refused - [ "$BINDABLE" -eq 1 ] && BIND_VERDICT=bindable - binding_check_result "$BIND_VERDICT" "$BIND_MECHANISM" "$BIND_REASON" "$BIND_RUN_HEAD" - [ "$BINDABLE" -eq 1 ] - exit $? - fi - if [ "$BINDABLE" -ne 1 ]; then - if [ "$BIND_PREPLAN_BLOCKED" -eq 1 ]; then - echo "error: No-Mistakes run predates the latest plan" >&2 - else - echo "error: No-Mistakes run does not match the latest plan" >&2 - fi - exit 2 - fi - [ -n "$BIND_GENERATION" ] || { echo "error: validation generation is missing" >&2; exit 2; } - printf 'validation_run_id=%s\nvalidation_run_path=%s\nvalidation_run_head=%s\nvalidation_run_generation=%s\nvalidation_run_binding=%s\n' \ - "$RUN_ID_INPUT" "$BIND_PATH" "$BIND_RUN_HEAD" "$BIND_GENERATION" "$BIND_MECHANISM" | append_meta_records \ - || { echo "error: could not bind the No-Mistakes run" >&2; exit 2; } - jq -cn --arg task "$ID" --arg run "$RUN_ID_INPUT" --arg path "$BIND_PATH" --arg head "$BIND_RUN_HEAD" \ - --arg binding "$BIND_MECHANISM" \ - '{schema:"fm-validation-run-binding.v1",task:$task,status:"bound",run:$run,path:$path,head:$head,binding:$binding}' - exit 0 -fi - -verify_mechanical_ready() { - local boundary worktree validated_head current_head validation_base new_receipts completion_files changed_file - [ "$(grep '^validation_path=' "$META" | tail -1 | cut -d= -f2- || true)" = receipts-mechanical ] \ - || { echo "error: latest plan does not use mechanical receipts" >&2; return 1; } - worktree=$(grep '^worktree=' "$META" | tail -1 | cut -d= -f2- || true) - validated_head=$(grep '^validation_head=' "$META" | tail -1 | cut -d= -f2- || true) - [ -n "$worktree" ] && [ -d "$worktree" ] || { echo "error: validation worktree is missing" >&2; return 1; } - validated_head=$(git -C "$worktree" rev-parse --verify "$validated_head^{commit}" 2>/dev/null) \ - || { echo "error: validated head is missing or invalid" >&2; return 1; } - current_head=$(git -C "$worktree" rev-parse --verify 'HEAD^{commit}' 2>/dev/null) \ - || { echo "error: current worktree head is unavailable" >&2; return 1; } - [ "$current_head" = "$validated_head" ] || { echo "error: current worktree head differs from the validated head; replan and revalidate" >&2; return 1; } - fm_worktree_is_clean "$worktree" || { echo "error: validation worktree is dirty; commit or remove all changes" >&2; return 1; } - boundary=$(grep '^validation_ledger_receipt_count=' "$META" | tail -1 | cut -d= -f2- || true) - case "$boundary" in ''|*[!0-9]*) echo "error: mechanical evidence boundary is missing" >&2; return 1 ;; esac - new_receipts="$TMP_ROOT/completion-new-receipts.jsonl" - tail -n "+$((boundary + 1))" "$LEDGER" > "$new_receipts" - validation_base=$(grep '^validation_base=' "$META" | tail -1 | cut -d= -f2- || true) - completion_files="$TMP_ROOT/completion-files" - git -C "$worktree" diff --no-ext-diff --no-renames --name-only "$validation_base..$validated_head" > "$completion_files" 2>/dev/null \ - && [ -s "$completion_files" ] || { echo "error: planned mechanical change files could not be observed" >&2; return 1; } - while IFS= read -r changed_file; do - [ -n "$changed_file" ] || continue - mechanical_evidence_covers_file "$new_receipts" "$changed_file" \ - || { echo "error: no applicable post-plan mechanical evidence was observed for $changed_file" >&2; return 1; } - done < "$completion_files" -} - -if [ "$ACTION" = mechanical-ready ]; then - verify_mechanical_ready || exit 2 - jq -cn --arg task "$ID" --argjson accepted_blocked "$ACCEPTED_BLOCKED_JSON" \ - '{schema:"fm-mechanical-readiness.v1",task:$task,status:"ready",accepted_blocked:$accepted_blocked}' - exit 0 -fi - -record_validation_completed() { - local started path generation published_generation completed completed_head completed_path completed_evidence completed_generation now worktree validation_base validated_head current_head completion_head expected_evidence observed pr pr_head branch boundary new_receipts run_id run_path run_generation run_out observed_id observed_head observed_head_full outcome run_status default_ref default_branch ci_state run_ready changed_file completion_files run_branch current_branch branch_sync_state run_head_matches_current restamp_accounted content_accounted run_binding expected_submitted done_claim ask_user_rc ask_user_report - VALIDATION_LOCK="$STATE/.$ID.validation-plan.lock" - if ! mkdir "$VALIDATION_LOCK" 2>/dev/null; then - VALIDATION_LOCK= - echo "error: validation metadata is locked by another planner" >&2 - return 1 - fi - started=$(grep '^validation_started_at=' "$META" | tail -1 | cut -d= -f2- || true) - path=$(grep '^validation_path=' "$META" | tail -1 | cut -d= -f2- || true) - generation=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) - worktree=$(grep '^worktree=' "$META" | tail -1 | cut -d= -f2- || true) - validation_base=$(grep '^validation_base=' "$META" | tail -1 | cut -d= -f2- || true) - validated_head=$(grep '^validation_head=' "$META" | tail -1 | cut -d= -f2- || true) - case "$started" in - ''|*[!0-9]*) release_validation_lock; echo "error: validation start timestamp is missing or invalid" >&2; return 1 ;; - esac - case "$path" in - receipts-mechanical) expected_evidence='pr-opened' ;; - full-no-mistakes) expected_evidence=no-mistakes-passed ;; - direct-PR) expected_evidence='pr-opened' ;; - local-only) expected_evidence='branch-ready' ;; - *) release_validation_lock; echo "error: validation path is missing or invalid" >&2; return 1 ;; - esac - [ "$TERMINAL_EVIDENCE" = "$expected_evidence" ] \ - || { release_validation_lock; echo "error: terminal evidence does not match validation path $path" >&2; return 1; } - # Completion-claim contract (worker-facing rules generated by - # bin/fm-brief.sh; the artifact check itself lives in - # bin/fm-classify-lib.sh): when the task's standing status is a done: claim, - # it must carry the delivery artifact this path requires - a canonical PR URL - # for the PR paths or "ready in branch" for local-only. A standing bare done: - # is a delivery claim with nothing to point at, so completion refuses it; a - # last line that is not a done: claim (working:, failed:, none) is untouched - # here and stays governed by the other gates. - done_claim=$(last_status_line "$STATE/$ID.status") - if [ "$(status_line_verb "${done_claim:-}")" = "done" ]; then - status_done_line_has_artifact "$done_claim" ship "$META_MODE" "$ID" \ - || { release_validation_lock; echo "error: the standing done: status line carries no delivery artifact for path $path" >&2; return 1; } - fi - [ -n "$worktree" ] && [ -d "$worktree" ] \ - || { release_validation_lock; echo "error: validation worktree is missing" >&2; return 1; } - validated_head=$(git -C "$worktree" rev-parse --verify "$validated_head^{commit}" 2>/dev/null) \ - || { release_validation_lock; echo "error: validated head is missing or invalid" >&2; return 1; } - if [ "$path" = full-no-mistakes ]; then - validation_base=$(git -C "$worktree" rev-parse --verify "$validation_base^{commit}" 2>/dev/null) \ - || { release_validation_lock; echo "error: validation base is missing or invalid" >&2; return 1; } - fi - current_head=$(git -C "$worktree" rev-parse --verify 'HEAD^{commit}' 2>/dev/null) \ - || { release_validation_lock; echo "error: current worktree head is unavailable" >&2; return 1; } - fm_worktree_is_clean "$worktree" \ - || { release_validation_lock; echo "error: validation worktree is dirty; commit or remove all changes" >&2; return 1; } - completion_head=$validated_head - if [ "$current_head" != "$validated_head" ] && [ "$path" != full-no-mistakes ]; then - printf 'validation_completed_at=\nvalidation_completed_head=\nvalidation_completed_path=\nvalidation_completed_evidence=\nvalidation_completed_generation=\n' | append_meta_records \ - || { release_validation_lock; echo "error: could not invalidate stale validation completion" >&2; return 1; } - release_validation_lock - echo "error: current worktree head differs from the validated head; replan and revalidate" >&2 - return 1 - fi - observed= - case "$path" in - receipts-mechanical) - verify_mechanical_ready \ - || { release_validation_lock; return 1; } - published_generation=$(grep '^validation_pr_published_generation=' "$META" | tail -1 | cut -d= -f2- || true) - [ "$published_generation" = "$generation" ] \ - || { release_validation_lock; echo "error: PR watcher was not published for the latest plan" >&2; return 1; } - pr=$(grep '^pr=' "$META" | tail -1 | cut -d= -f2- || true) - pr_head=$(grep '^pr_head=' "$META" | tail -1 | cut -d= -f2- || true) - case "$pr" in - https://github.com/*) - [ "$pr_head" = "$validated_head" ] \ - || { release_validation_lock; echo "error: GitHub PR head is missing or not bound to the validated head" >&2; return 1; } - ;; - https://*) ;; - *) release_validation_lock; echo "error: canonical PR metadata is missing" >&2; return 1 ;; - esac - observed=post-plan-mechanical-receipt-and-pr - ;; - full-no-mistakes) - run_id=$(grep '^validation_run_id=' "$META" | tail -1 | cut -d= -f2- || true) - run_path=$(grep '^validation_run_path=' "$META" | tail -1 | cut -d= -f2- || true) - run_generation=$(grep '^validation_run_generation=' "$META" | tail -1 | cut -d= -f2- || true) - observed_head=$(grep '^validation_run_head=' "$META" | tail -1 | cut -d= -f2- || true) - [ -n "$run_id" ] && [ "$run_path" = "$path" ] && [ "$run_generation" = "$generation" ] && [ -n "$observed_head" ] \ - || { release_validation_lock; echo "error: no No-Mistakes run is bound to the latest plan" >&2; return 1; } - run_out=$(fm_nm_run_checked "$worktree" "$NM_TIMEOUT" axi status --run "$run_id") \ - || { release_validation_lock; echo "error: bound No-Mistakes run could not be observed" >&2; return 1; } - observed_id=$(fm_nm_field "$run_out" id) - observed_head=$(fm_nm_field "$run_out" head) - outcome=$(fm_nm_field "$run_out" outcome) - run_status=$(fm_nm_field "$run_out" status) - run_ready=0 - if [ "$outcome" = passed ] || [ "$outcome" = passed-with-override ] || [ "$outcome" = checks-passed ] || [ "$run_status" = checks-passed ]; then - run_ready=1 - elif [ "$run_status" = ci ] || [ "$run_status" = running ]; then - ci_state=$(fm_nm_ci_checks_state "$worktree" "$NM_TIMEOUT" "$run_id") - [ "$ci_state" != green ] || run_ready=1 - fi - observed_head_full=$(fm_nm_resolve_head "$worktree" "$observed_head" || true) - if [ -z "$observed_head_full" ] || [ "$observed_id" != "$run_id" ] || [ "$run_ready" -ne 1 ]; then - release_validation_lock - echo "error: bound No-Mistakes run did not pass checks at the exact validated head" >&2 - return 1 - fi - # The run must account for the current head itself, both heads must be - # faithful restamps of the validated chain from the recorded base, or the - # run's reported head and the checked-out head must carry identical - # trees: a mid-run rebase onto a newer base breaks every ancestry shape - # while the bound run still validates exactly the shipping content. Tree - # equality never stands alone - the ownership gate below still requires - # active pipeline ownership or a terminal passed run. - restamp_accounted=0 - content_accounted=0 - run_binding=$(grep '^validation_run_binding=' "$META" | tail -1 | cut -d= -f2- || true) - if { [ "$run_binding" = content-tree ] || [ "$current_head" != "$validated_head" ]; } \ - && fm_nm_commits_share_tree "$worktree" "$observed_head_full" "$current_head"; then - content_accounted=1 - fi - if [ "$observed_head_full" = "$current_head" ]; then - run_head_matches_current=1 - elif fm_nm_head_is_faithful_restamp "$worktree" "$validation_base" "$validated_head" "$observed_head_full" \ - && fm_nm_head_is_faithful_restamp "$worktree" "$validation_base" "$validated_head" "$current_head"; then - run_head_matches_current=1 - restamp_accounted=1 - elif fm_nm_commits_share_tree "$worktree" "$observed_head_full" "$current_head"; then - run_head_matches_current=1 - content_accounted=1 - else - run_head_matches_current=0 - fi - [ "$run_head_matches_current" -eq 1 ] \ - || { release_validation_lock; echo "error: bound No-Mistakes run head does not account for the current worktree content" >&2; return 1; } - if [ "$current_head" != "$validated_head" ]; then - if [ "$content_accounted" -eq 0 ] \ - && ! fm_nm_head_is_accounted "$worktree" "$validation_base" "$validated_head" "$current_head"; then - if fm_nm_commits_share_tree "$worktree" "$observed_head_full" "$current_head"; then - content_accounted=1 - else - release_validation_lock - echo "error: current head neither descends from nor reproduces the implementation head" >&2 - return 1 - fi - fi - completion_head=$current_head - fi - if [ "$current_head" != "$validated_head" ] || [ "$restamp_accounted" -eq 1 ] || [ "$content_accounted" -eq 1 ]; then - run_branch=$(fm_nm_field "$run_out" branch) - current_branch=$(git -C "$worktree" symbolic-ref --quiet --short HEAD 2>/dev/null || true) - if [ -z "$current_branch" ] || ! fm_nm_branch_matches_worktree "$worktree" "$run_branch"; then - release_validation_lock - echo "error: pipeline run branch is not the current worktree branch" >&2 - return 1 - fi - # The advance is authoritative only while the run is ACTIVE and the - # pipeline owns the branch, or once the run has reached a terminal PASSED - # state and released the branch. Active ownership is proven by the shared - # fm_nm_run_branch_ownership predicate: a branch_sync state of - # pipeline_owned, either directly in the axi status output or in `axi - # sync --check` for current no-mistakes, or the fully evidenced - # synchronized state once the pushed-back head converged and the run - # only monitors its PR (bin/fm-nm-run-lib.sh). A content-accounted - # advance leaves the submitted anchor open because a run bound through - # content identity may have submitted before the plan recorded its head - # or submitted the pre-rebase chain. - if fm_nm_run_is_active "$run_out"; then - expected_submitted=$validated_head - [ "$content_accounted" -eq 1 ] && expected_submitted= - branch_sync_state=$(fm_nm_run_branch_ownership "$worktree" "$NM_TIMEOUT" \ - "$run_out" "$run_id" "$expected_submitted" "$observed_head_full" || true) - if [ -z "$branch_sync_state" ]; then - release_validation_lock - if [ "$content_accounted" -eq 1 ]; then - echo "error: content-identical head lacks authoritative pipeline ownership" >&2 - elif [ "$restamp_accounted" -eq 1 ]; then - echo "error: accepted restamp lacks authoritative pipeline ownership" >&2 - else - echo "error: current head advance lacks authoritative pipeline ownership" >&2 - fi - return 1 - fi - else - if ! fm_nm_run_is_terminal_passed "$run_out"; then - release_validation_lock - if [ "$content_accounted" -eq 1 ]; then - echo "error: content-identical head lacks a passing bound pipeline run" >&2 - elif [ "$restamp_accounted" -eq 1 ]; then - echo "error: accepted restamp lacks a passing bound pipeline run" >&2 - else - echo "error: current head advanced without a passing bound pipeline run" >&2 - fi - return 1 - fi - fi - fi - # Apply the same decision-evidence check as PR-ready before recording - # completion (contract: bin/fm-nm-run-lib.sh). - ask_user_rc=0 - ask_user_report=$(fm_nm_ask_user_decisions "$worktree" "$NM_TIMEOUT" "$run_id" "$STATE/$ID.status") \ - || ask_user_rc=$? - if [ "$ask_user_rc" -ne 0 ]; then - release_validation_lock - if [ "$ask_user_rc" -eq 1 ]; then - echo "error: bound No-Mistakes run $run_id resolved ask-user findings without matching firstmate decisions" >&2 - printf '%s\n' "$ask_user_report" >&2 - echo "error: firstmate must record one resolved [key=nm-$run_id-] line per decision event in state/$ID.status" >&2 - else - echo "error: bound No-Mistakes run $run_id ask-user decision evidence could not be read" >&2 - fi - return 1 - fi - observed="bound-matching-no-mistakes-run" - ;; - direct-PR) - published_generation=$(grep '^validation_pr_published_generation=' "$META" | tail -1 | cut -d= -f2- || true) - [ "$published_generation" = "$generation" ] \ - || { release_validation_lock; echo "error: PR watcher was not published for the latest plan" >&2; return 1; } - pr=$(grep '^pr=' "$META" | tail -1 | cut -d= -f2- || true) - pr_head=$(grep '^pr_head=' "$META" | tail -1 | cut -d= -f2- || true) - case "$pr" in - https://github.com/*) - [ "$pr_head" = "$validated_head" ] \ - || { release_validation_lock; echo "error: GitHub PR head is missing or not bound to the validated head" >&2; return 1; } - observed=canonical-github-pr-head - ;; - https://*) observed=canonical-non-github-pr ;; - *) release_validation_lock; echo "error: canonical PR metadata is missing" >&2; return 1 ;; - esac - ;; - local-only) - branch=$(git -C "$worktree" symbolic-ref --quiet --short HEAD 2>/dev/null || true) - [ "$branch" = "fm/$ID" ] \ - || { release_validation_lock; echo "error: local-only branch is not ready" >&2; return 1; } - default_branch=$("$SCRIPT_DIR/fm-local-default.sh" "$worktree") \ - || { release_validation_lock; echo "error: authoritative local default branch is missing" >&2; return 1; } - default_ref="refs/heads/$default_branch" - if [ -z "$default_ref" ] \ - || ! git -C "$worktree" merge-base --is-ancestor "$default_ref" "$validated_head" 2>/dev/null; then - release_validation_lock - echo "error: local-only branch is not fast-forward ready" >&2 - return 1 - fi - observed=clean-ready-branch - ;; - esac - completed=$(grep '^validation_completed_at=' "$META" | tail -1 | cut -d= -f2- || true) - completed_head=$(grep '^validation_completed_head=' "$META" | tail -1 | cut -d= -f2- || true) - completed_path=$(grep '^validation_completed_path=' "$META" | tail -1 | cut -d= -f2- || true) - completed_evidence=$(grep '^validation_completed_evidence=' "$META" | tail -1 | cut -d= -f2- || true) - completed_generation=$(grep '^validation_completed_generation=' "$META" | tail -1 | cut -d= -f2- || true) - if [ -n "$completed_head" ]; then - [ -n "$completed" ] \ - || { release_validation_lock; echo "error: validation completion timestamp is missing" >&2; return 1; } - case "$completed" in - *[!0-9]*) release_validation_lock; echo "error: validation completion timestamp is invalid" >&2; return 1 ;; - esac - completed_head=$(git -C "$worktree" rev-parse --verify "$completed_head^{commit}" 2>/dev/null) \ - || { release_validation_lock; echo "error: validation completed head is invalid" >&2; return 1; } - fi - if [ "$completed_head:$completed_path:$completed_evidence:$completed_generation" != "$completion_head:$path:$observed:$generation" ]; then - now=$(date +%s) - printf 'validation_completed_at=%s\nvalidation_completed_head=%s\nvalidation_completed_path=%s\nvalidation_completed_evidence=%s\nvalidation_completed_generation=%s\n' \ - "$now" "$completion_head" "$path" "$observed" "$generation" | append_meta_records \ - || { release_validation_lock; echo "error: could not record validation completion" >&2; return 1; } - completed=$now - completed_head=$completion_head - fi - release_validation_lock - VALIDATION_COMPLETED=$completed - VALIDATION_COMPLETED_HEAD=$completed_head - VALIDATION_COMPLETED_PATH=$path - VALIDATION_COMPLETED_EVIDENCE=$observed -} - -if [ "$ACTION" = complete ]; then - record_validation_completed || exit 2 - jq -cn --arg task "$ID" --argjson completed_at "$VALIDATION_COMPLETED" --arg completed_head "$VALIDATION_COMPLETED_HEAD" \ - --arg path "$VALIDATION_COMPLETED_PATH" --arg evidence "$VALIDATION_COMPLETED_EVIDENCE" \ - --argjson accepted_blocked "$ACCEPTED_BLOCKED_JSON" \ - '{schema:"fm-validation-completion.v1",task:$task,status:"completed",completed_at:$completed_at,completed_head:$completed_head,path:$path,evidence:$evidence,accepted_blocked:$accepted_blocked}' - exit 0 -fi - -[ -f "$META" ] && [ ! -L "$META" ] \ - || { echo "error: task metadata is missing or unsafe: $META" >&2; exit 2; } -WORKTREE=$(grep '^worktree=' "$META" 2>/dev/null | tail -1 | cut -d= -f2- || true) -[ -n "$WORKTREE" ] && [ -d "$WORKTREE" ] \ - || { echo "error: validation worktree is missing" >&2; exit 2; } -if ! fm_worktree_is_clean "$WORKTREE"; then - echo "error: validation worktree is dirty; commit or remove all changes" >&2 - exit 2 -fi - -BASE= -HEAD= -DIFF_AVAILABLE=0 -DIFF_FILES=0 -DIFF_LINES=0 -HAS_BINARY=0 -HAS_SPECIAL_MODE=0 -LOW_PATH=1 -LOW_STRUCTURE=0 -NUMSTAT="$TMP_ROOT/numstat" -NAMES="$TMP_ROOT/names" -: > "$NUMSTAT" -: > "$NAMES" - -resolve_diff() { - local requested_base authoritative_base origin_head - [ -n "$WORKTREE" ] && [ -d "$WORKTREE" ] && git -C "$WORKTREE" rev-parse --git-dir >/dev/null 2>&1 || return 1 - fm_worktree_is_clean "$WORKTREE" || return 1 - HEAD=$(git -C "$WORKTREE" rev-parse --verify 'HEAD^{commit}' 2>/dev/null) || return 1 - origin_head=$(git -C "$WORKTREE" symbolic-ref --quiet refs/remotes/origin/HEAD 2>/dev/null || true) - if [ -n "$origin_head" ]; then - authoritative_base=$(git -C "$WORKTREE" merge-base HEAD "$origin_head" 2>/dev/null) || return 1 - else - authoritative_base=$(git -C "$WORKTREE" merge-base HEAD main 2>/dev/null \ - || git -C "$WORKTREE" merge-base HEAD master 2>/dev/null) || return 1 - fi - BASE=$authoritative_base - if [ -n "$BASE_INPUT" ]; then - requested_base=$(git -C "$WORKTREE" rev-parse --verify "$BASE_INPUT^{commit}" 2>/dev/null) || return 1 - [ "$requested_base" = "$authoritative_base" ] || return 1 - fi - git -C "$WORKTREE" merge-base --is-ancestor "$BASE" "$HEAD" 2>/dev/null || return 1 - git -C "$WORKTREE" diff --no-ext-diff --no-renames --numstat "$BASE..$HEAD" > "$NUMSTAT" 2>/dev/null || return 1 - git -C "$WORKTREE" diff --no-ext-diff --no-renames --name-only "$BASE..$HEAD" > "$NAMES" 2>/dev/null || return 1 - DIFF_SUMMARY="$TMP_ROOT/diff-summary" - git -C "$WORKTREE" diff --no-ext-diff --no-renames --summary "$BASE..$HEAD" > "$DIFF_SUMMARY" 2>/dev/null \ - || return 1 - if grep -Eq '(mode change|mode (100755|120000|160000))' "$DIFF_SUMMARY"; then - HAS_SPECIAL_MODE=1 - fi - DIFF_AVAILABLE=1 -} - -resolve_diff \ - || { echo "error: authoritative validation base and diff could not be resolved" >&2; exit 2; } -IMPLEMENTATION_COMPLETED=$(grep '^implementation_completed_at=' "$META" | tail -1 | cut -d= -f2- || true) -IMPLEMENTATION_HEAD=$(grep '^implementation_completed_head=' "$META" | tail -1 | cut -d= -f2- || true) -case "$IMPLEMENTATION_COMPLETED" in - ''|*[!0-9]*) echo "error: record implementation completion before planning" >&2; exit 2 ;; -esac -[ -n "$HEAD" ] && [ "$IMPLEMENTATION_HEAD" = "$HEAD" ] \ - || { echo "error: implementation completion is not bound to the current head" >&2; exit 2; } -if [ "$DIFF_AVAILABLE" -eq 1 ]; then - while IFS=$'\t' read -r added deleted path; do - [ -n "$path" ] || continue - DIFF_FILES=$((DIFF_FILES + 1)) - case "$added:$deleted" in - *-*) HAS_BINARY=1 ;; - *) DIFF_LINES=$((DIFF_LINES + added + deleted)) ;; - esac - case "$path" in - CHANGELOG.md) ;; - *) LOW_PATH=0 ;; - esac - done < "$NUMSTAT" -fi - -if [ "$DIFF_AVAILABLE" -eq 1 ] && [ "$LOW_PATH" -eq 1 ]; then - LOW_PATCH="$TMP_ROOT/low-prose.patch" - if git -C "$WORKTREE" diff --no-ext-diff --no-renames --unified=0 "$BASE..$HEAD" -- CHANGELOG.md > "$LOW_PATCH" 2>/dev/null \ - && awk ' - BEGIN { removed=0; added=0; old_bytes=""; new_bytes=""; bad=0 } - /^\+\+\+ / || /^--- / || /^@@/ || /^diff --git / || /^index / { next } - /^-/ { - line=substr($0, 2) - if (line !~ /^[[:alnum:]][[:alnum:][:space:].,;:!?()"'"'"'-]*$/) { bad=1; next } - removed++ - gsub(/[[:space:]]/, "", line) - old_bytes=old_bytes line - next - } - /^\+/ { - line=substr($0, 2) - if (line !~ /^[[:alnum:]][[:alnum:][:space:].,;:!?()"'"'"'-]*$/) { bad=1; next } - added++ - gsub(/[[:space:]]/, "", line) - new_bytes=new_bytes line - next - } - END { - exit(!bad && removed > 0 && added > 0 && old_bytes != "" && old_bytes == new_bytes ? 0 : 1) - } - ' "$LOW_PATCH"; then - LOW_STRUCTURE=1 - fi -fi - -MECHANICAL_PROOF=1 -if [ "$DIFF_AVAILABLE" -eq 1 ]; then - while IFS= read -r changed_file; do - [ -n "$changed_file" ] || continue - if ! mechanical_evidence_covers_file "$LEDGER" "$changed_file"; then - MECHANICAL_PROOF=0 - break - fi - done < "$NAMES" -else - MECHANICAL_PROOF=0 -fi - -TIER=high -REASON=uncertain-input -if [ "$DIFF_AVAILABLE" -eq 0 ] || [ "$DIFF_FILES" -eq 0 ]; then - TIER=high - REASON=unreadable-or-empty-diff -elif [ "$HAS_SPECIAL_MODE" -eq 1 ]; then - TIER=high - REASON=special-file-or-mode-change -elif [ "$HAS_BINARY" -eq 1 ]; then - TIER=high - REASON=binary-change -elif [ "$DIFF_FILES" -gt 8 ] || [ "$DIFF_LINES" -gt 400 ]; then - TIER=high - REASON=broad-change -elif [ "$LOW_PATH" -eq 1 ] && [ "$LOW_STRUCTURE" -eq 1 ] && [ "$DIFF_FILES" -eq 1 ] && [ "$DIFF_LINES" -le 4 ] \ - && [ "$MECHANICAL_PROOF" -eq 1 ]; then - TIER=low - REASON=non-authoritative-prose -else - TIER=high - REASON=default-high -fi - -case "$MODE:$TIER" in - direct-PR:*) VALIDATION_PATH=direct-PR ;; - local-only:*) VALIDATION_PATH=local-only ;; - no-mistakes:low) VALIDATION_PATH=receipts-mechanical ;; - no-mistakes:high) VALIDATION_PATH=full-no-mistakes ;; -esac - -# A plan identical to the latest one (same resolved base and head) cannot add -# evidence, so it must never silently clear a bound No-Mistakes run: the -# preplan boundary would then record that run as predating the plan and refuse -# its binding, orphaning live validation behind a forced re-review. A replan -# over genuinely changed content still clears the stale binding because the -# bound run validated a different head. -if [ "$VALIDATION_PATH" = full-no-mistakes ]; then - PREVIOUS_GENERATION=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) - BOUND_RUN=$(grep '^validation_run_id=' "$META" | tail -1 | cut -d= -f2- || true) - BOUND_RUN_GENERATION=$(grep '^validation_run_generation=' "$META" | tail -1 | cut -d= -f2- || true) - PREVIOUS_HEAD=$(grep '^validation_head=' "$META" | tail -1 | cut -d= -f2- || true) - PREVIOUS_BASE=$(grep '^validation_base=' "$META" | tail -1 | cut -d= -f2- || true) - if [ -n "$BOUND_RUN" ] && [ -n "$PREVIOUS_GENERATION" ] \ - && [ "$BOUND_RUN_GENERATION" = "$PREVIOUS_GENERATION" ] \ - && [ "$PREVIOUS_HEAD" = "$HEAD" ] && [ "$PREVIOUS_BASE" = "$BASE" ]; then - echo "error: run $BOUND_RUN is already bound to an identical validation plan; a same-content replan cannot add evidence - bind a new run to the current plan or advance the head before replanning" >&2 - exit 2 - fi -fi - -write_meta_record() { # - local pass=$1 started previous_generation - VALIDATION_LOCK="$STATE/.$ID.validation-plan.lock" - if ! mkdir "$VALIDATION_LOCK" 2>/dev/null; then - VALIDATION_LOCK= - echo "error: validation metadata is locked by another planner: $STATE/.$ID.validation-plan.lock" >&2 - return 1 - fi - started=$(grep '^validation_started_at=' "$META" | tail -1 | cut -d= -f2- || true) - previous_generation=$(grep '^validation_generation=' "$META" | tail -1 | cut -d= -f2- || true) - case "$started" in - '') ;; - *[!0-9]*) release_validation_lock; echo "error: validation start timestamp is invalid" >&2; return 1 ;; - esac - if ! { - printf 'validation_generation=%s\n' "$PLAN_GENERATION" - printf 'validation_tier=%s\n' "$TIER" - printf 'validation_path=%s\n' "$VALIDATION_PATH" - printf 'validation_reason=%s\n' "$REASON" - printf 'validation_base=%s\n' "$BASE" - printf 'validation_head=%s\n' "$HEAD" - printf 'validation_diff_files=%s\n' "$DIFF_FILES" - printf 'validation_diff_lines=%s\n' "$DIFF_LINES" - printf 'validation_pass=%s\n' "$pass" - [ "$started" = "$IMPLEMENTATION_COMPLETED" ] || printf 'validation_started_at=%s\n' "$IMPLEMENTATION_COMPLETED" - printf 'validation_ledger_receipt_count=%s\n' "$RECEIPT_COUNT" - printf 'validation_preplan_run_id=%s\n' "${PREPLAN_RUN_ID:-}" - printf 'validation_pr_published_generation=\n' - if [ -n "$previous_generation" ]; then - printf 'validation_run_id=\nvalidation_run_path=\nvalidation_run_head=\nvalidation_run_generation=\n' - printf 'validation_completed_at=\nvalidation_completed_head=\nvalidation_completed_path=\nvalidation_completed_evidence=\nvalidation_completed_generation=\n' - fi - } | append_meta_records; then - release_validation_lock - echo "error: could not append validation metadata: $META" >&2 - return 1 - fi - release_validation_lock -} - -PLAN_GENERATION=$(od -An -N16 -tx1 /dev/urandom 2>/dev/null | tr -d ' \n') -[ "${#PLAN_GENERATION}" -eq 32 ] || { echo "error: validation generation could not be created" >&2; exit 2; } -PREPLAN_RUN_ID= -if [ "$VALIDATION_PATH" = full-no-mistakes ]; then - PREPLAN_OUT=$(fm_nm_run_checked "$WORKTREE" "$NM_TIMEOUT" axi status) \ - || { echo "error: pre-plan No-Mistakes boundary could not be observed" >&2; exit 2; } - PREPLAN_RUN_ID=$(fm_nm_field "$PREPLAN_OUT" id) -fi -write_meta_record initial -RECEIPT_COMMAND= -MECHANICAL_COMMAND= -PUSH_COMMAND= -PR_COMMAND= -DONE_STATUS= -REGISTER_COMMAND= -if [ "$VALIDATION_PATH" = receipts-mechanical ]; then - RECEIPT_COMMAND="bin/fm-receipt.sh $ID --outcome success --file " - MECHANICAL_COMMAND="bin/fm-receipt-check.sh $ID --mechanical-ready" - PUSH_COMMAND="git push -u origin fm/$ID" - PR_COMMAND="gh-axi pr create " - DONE_STATUS="done: PR " - REGISTER_COMMAND="bin/fm-pr-check.sh $ID " -fi -jq -cn --arg task "$ID" --arg mode "$MODE" --arg tier "$TIER" --arg path "$VALIDATION_PATH" --arg reason "$REASON" \ - --arg base "$BASE" --arg head "$HEAD" --arg generation "$PLAN_GENERATION" \ - --arg receipt_command "$RECEIPT_COMMAND" --arg mechanical_command "$MECHANICAL_COMMAND" \ - --arg push_command "$PUSH_COMMAND" --arg pr_command "$PR_COMMAND" --arg done_status "$DONE_STATUS" \ - --arg register_command "$REGISTER_COMMAND" \ - --argjson diff_files "$DIFF_FILES" --argjson diff_lines "$DIFF_LINES" \ - --argjson accepted_blocked "$ACCEPTED_BLOCKED_JSON" \ - '{schema:"fm-validation-plan.v1",task:$task,status:"planned",mode:$mode,tier:$tier,path:$path,reason:$reason,base:$base,head:$head,generation:$generation,diff_files:$diff_files,diff_lines:$diff_lines,accepted_blocked:$accepted_blocked} - + (if $path == "receipts-mechanical" then {receipt_command:$receipt_command,mechanical_command:$mechanical_command,push_command:$push_command,pr_command:$pr_command,done_status:$done_status,register_command:$register_command} else {} end) - + (if ($accepted_blocked | length) > 0 then {accepted_blocked_note:"state these captain-accepted blocked criteria and their exception references plainly in the PR description; firstmate never auto-merges a task with any accepted-blocked criterion"} else {} end)' + '{schema:"fm-evidence-check.v2",task:$task,kind:"ship",status:$status,required:$required,evidenced:$accounting.evidenced,accepted_blocked:$accounting.accepted_blocked,missing:$accounting.missing,invalid:$invalid}' +exit "$CHECK_RC" diff --git a/bin/fm-receipt-schema.sh b/bin/fm-receipt-schema.sh index 248c9c98601..72e65911667 100755 --- a/bin/fm-receipt-schema.sh +++ b/bin/fm-receipt-schema.sh @@ -4,12 +4,13 @@ # Usage: fm-receipt-schema.sh # # The input must be one JSON object with required criterion, type, outcome, -# summary, and result string fields; optional command, artifact, file, and head -# strings; no unknown keys; type set to test, build, lint, typecheck, api, -# browser, manual, or review; outcome set to success, failure, negative, zero, -# skipped, empty, placeholder, weak, passed, failed, or accepted-blocked; a -# non-whitespace captain_exception string present exactly when the outcome is -# accepted-blocked; and a 40- or 64-hex head when head is present. +# summary, and result string fields; optional command, artifact, and file +# strings; an optional 40- or 64-hex head string that older ledgers carry and no +# current writer emits; no unknown keys; type set to test, build, lint, +# typecheck, api, browser, manual, or review; outcome set to success, failure, +# negative, zero, skipped, empty, placeholder, weak, passed, failed, or +# accepted-blocked; and a non-whitespace captain_exception string present +# exactly when the outcome is accepted-blocked. set -eu usage() { diff --git a/bin/fm-receipt-store.sh b/bin/fm-receipt-store.sh index 89e7f635f7e..335be8582d8 100755 --- a/bin/fm-receipt-store.sh +++ b/bin/fm-receipt-store.sh @@ -6,7 +6,6 @@ # fm-receipt-store.sh hold # fm-receipt-store.sh append # fm-receipt-store.sh meta-read -# fm-receipt-store.sh meta-append # fm-receipt-store.sh meta-replace # fm-receipt-store.sh promote [ ...] # @@ -18,6 +17,9 @@ # one line before exiting with the same status. # append validates the pinned ship brief and criterion, then appends the compact # JSON payload from FM_RECEIPT_PAYLOAD under an exclusive ledger lock. +# meta-read copies the pinned state/.meta; meta-replace replaces the +# FM_RECEIPT_META_REPLACE_KEYS records atomically after confirming the file still +# equals expected-meta, so a PR publication cannot overwrite a concurrent change. # promote holds the exclusive task lock across the transaction child's documented # phases, durably commits task and state replacements, recovers identity-bound # unfinished work, and retains the committed record through retirement. @@ -59,7 +61,7 @@ ID=$1 MODE=$2 shift 2 case "$MODE" in - meta-read|meta-append|meta-replace) + meta-read|meta-replace) case "$ID" in ''|.*|*[!A-Za-z0-9._-]*) echo "error: invalid task id: $ID" >&2; exit 2 ;; esac ;; *) @@ -67,7 +69,7 @@ case "$MODE" in ;; esac case "$MODE:$#" in - scaffold:0|hold:5|append:2|meta-read:1|meta-append:3|meta-replace:3) ;; + scaffold:0|hold:5|append:2|meta-read:1|meta-replace:3) ;; promote:0) usage >&2; exit 2 ;; promote:*) ;; *) usage >&2; exit 2 ;; @@ -235,30 +237,27 @@ sub run_metadata_operation { or refuse("metadata update input could not be read"); refuse("task metadata changed during validation") unless $current_text eq $expected_text; refuse("metadata records are empty") unless length($records_text); - my $new_text = $current_text; - if ($mode eq "meta-replace") { - my $keys_text = $ENV{FM_RECEIPT_META_REPLACE_KEYS} // ""; - my @keys = split(/,/, $keys_text, -1); - refuse("metadata replacement keys are missing") unless @keys; - my %replace; - for my $key (@keys) { - refuse("metadata replacement key is invalid") unless $key =~ /\A[A-Za-z][A-Za-z0-9_]*\z/; - refuse("metadata replacement key is duplicated") if $replace{$key}++; - } - my %recorded; - for my $line (split(/\n/, $records_text, -1)) { - next unless length($line); - my ($key) = $line =~ /\A([A-Za-z][A-Za-z0-9_]*)=/; - refuse("metadata replacement record is invalid") unless defined($key) && $replace{$key}; - refuse("metadata replacement record is duplicated") if $recorded{$key}++; - } - my @kept = grep { - my ($key) = /\A([A-Za-z][A-Za-z0-9_]*)=/; - !defined($key) || !$replace{$key} - } split(/\n/, $current_text, -1); - $new_text = join("\n", @kept); - $new_text =~ s/\n*\z//; + my $keys_text = $ENV{FM_RECEIPT_META_REPLACE_KEYS} // ""; + my @keys = split(/,/, $keys_text, -1); + refuse("metadata replacement keys are missing") unless @keys; + my %replace; + for my $key (@keys) { + refuse("metadata replacement key is invalid") unless $key =~ /\A[A-Za-z][A-Za-z0-9_]*\z/; + refuse("metadata replacement key is duplicated") if $replace{$key}++; + } + my %recorded; + for my $line (split(/\n/, $records_text, -1)) { + next unless length($line); + my ($key) = $line =~ /\A([A-Za-z][A-Za-z0-9_]*)=/; + refuse("metadata replacement record is invalid") unless defined($key) && $replace{$key}; + refuse("metadata replacement record is duplicated") if $recorded{$key}++; } + my @kept = grep { + my ($key) = /\A([A-Za-z][A-Za-z0-9_]*)=/; + !defined($key) || !$replace{$key} + } split(/\n/, $current_text, -1); + my $new_text = join("\n", @kept); + $new_text =~ s/\n*\z//; $new_text .= "\n" if length($new_text) && $new_text !~ /\n\z/; $new_text .= $records_text; $new_text .= "\n" if $new_text !~ /\n\z/; @@ -319,8 +318,7 @@ sub retire_promotion_task_artifacts { } my $task_name = $ENV{FM_RECEIPT_STORE_ID}; -if ($ENV{FM_RECEIPT_STORE_MODE} eq "meta-read" || $ENV{FM_RECEIPT_STORE_MODE} eq "meta-append" - || $ENV{FM_RECEIPT_STORE_MODE} eq "meta-replace") { +if ($ENV{FM_RECEIPT_STORE_MODE} eq "meta-read" || $ENV{FM_RECEIPT_STORE_MODE} eq "meta-replace") { run_metadata_operation($ENV{FM_RECEIPT_STORE_STATE}, "$task_name.meta", $ENV{FM_RECEIPT_STORE_MODE}, @ARGV); exit 0; } @@ -679,40 +677,8 @@ open(my $criterion_parser, "|-", $parser, "--parse-criteria", "-", "--require", print {$criterion_parser} $brief_text or refuse("task brief could not reach the acceptance-criterion parser"); close($criterion_parser) or refuse("criterion is not declared by a valid ship brief: $criterion"); -chdir($state) or refuse("pinned state directory could not be re-entered for receipt binding"); -my @receipt_meta_identity = lstat($meta_name); -if (!@receipt_meta_identity || S_ISLNK($receipt_meta_identity[2]) - || $receipt_meta_identity[0] != $meta_identity[0] - || $receipt_meta_identity[1] != $meta_identity[1]) { - close($meta) or refuse("superseded task metadata could not be closed"); - sysopen($meta, $meta_name, O_RDONLY | O_NOFOLLOW) - or refuse("current task metadata is missing or unsafe"); - @meta_identity = stat($meta); - refuse("current task metadata must be a single-link regular file") unless @meta_identity - && S_ISREG($meta_identity[2]) && $meta_identity[3] == 1; - @receipt_meta_identity = lstat($meta_name); - refuse("current task metadata identity changed during receipt binding") unless @receipt_meta_identity - && !S_ISLNK($receipt_meta_identity[2]) - && $receipt_meta_identity[0] == $meta_identity[0] - && $receipt_meta_identity[1] == $meta_identity[1]; -} -chdir($task) or refuse("pinned task directory could not be re-entered after receipt binding"); -sysseek($meta, 0, 0) or refuse("task metadata could not be rewound for receipt binding"); -local $/; -my $meta_text = <$meta>; -defined($meta_text) or refuse("task metadata could not be read for receipt binding"); -my @worktrees = ($meta_text =~ /^worktree=(.*)$/mg); my $payload = eval { decode_json($ENV{FM_RECEIPT_PAYLOAD}) }; refuse("evidence receipt payload is invalid") unless defined($payload) && ref($payload) eq "HASH"; -if (@worktrees == 1 && length($worktrees[0])) { - if (open(my $git_head, "-|", "git", "-C", $worktrees[0], "rev-parse", "--verify", "HEAD^{commit}")) { - my $head = <$git_head>; - if (close($git_head) && defined($head)) { - $head =~ s/\r?\n\z//; - $payload->{head} = $head if $head =~ /^(?:[0-9a-f]{40}|[0-9a-f]{64})$/; - } - } -} my $payload_text = encode_json($payload); my $record = "$payload_text\n"; sysopen(my $random, "/dev/urandom", O_RDONLY | O_NOFOLLOW) diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 6f2ec166b6b..c577d3d053c 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -19,7 +19,10 @@ # no-mistakes-prod-only is a registry policy rather than a task mode and is # refused as a flag value. # A ship relaunch recovers --mode and --yolo from the recorded task metadata -# when the caller does not repeat them, and a non-secondmate OMP relaunch also +# when the caller does not repeat them, carries the delivery identity records +# pr=, pr_head=, and nm_run_id= (written only by bin/fm-pr-check.sh) into the +# replacement record so a restart mid-handoff keeps the registered PR and +# No-Mistakes run, and a non-secondmate OMP relaunch also # recovers --prewalk-into and --allow-project-omp-extensions so the replacement # worker keeps the prior launch intent. # A relaunch validates the recorded endpoint before reusing it. A @@ -4974,6 +4977,15 @@ fi echo "model=${MODEL:-default}" echo "effort=${EFFORT:-default}" echo "spawn_gen=$SPAWN_GEN" + # Delivery identity survives a relaunch: fm-pr-check.sh is the only writer of + # these records and a restarted worker cannot recreate them, so the done and + # PR-ready gates would otherwise never be satisfiable after a restart. + if [ "$RELAUNCH" -eq 1 ]; then + for delivery_key in pr pr_head nm_run_id; do + delivery_value=$(fm_meta_get "$RELAUNCH_META" "$delivery_key") + [ -z "$delivery_value" ] || echo "$delivery_key=$delivery_value" + done + fi # The relaunch transaction id lets fm-control classify a post-publish # failure as "new record published" rather than "replacement never # launched"; only a relaunch under fm-control writes it. diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index f7836ff7e05..eaae244add7 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -206,7 +206,7 @@ family_for_basename() { fm-teardown-endpoint-safety.test.sh) printf '%s\n' backend-dispatch ;; - fm-check-unregister.test.sh|fm-main-ci-watch.test.sh|fm-pr-check-security.test.sh|fm-pr-merge.test.sh|\ + fm-check-unregister.test.sh|fm-main-ci-watch.test.sh|fm-pr-check-handoff.test.sh|fm-pr-check-security.test.sh|fm-pr-merge.test.sh|\ fm-review-diff.test.sh|fm-teardown.test.sh|fm-x-mode.test.sh|fm-ext-bridge.test.sh) printf '%s\n' pr-forge ;; @@ -455,13 +455,14 @@ tests/fm-pending-reply.test.sh 19949 tests/fm-pi-compatible-family.test.sh 56 tests/fm-pi-primary-live-e2e.test.sh 21 tests/fm-pi-watch-extension.test.sh 22031 +tests/fm-pr-check-handoff.test.sh 42228 tests/fm-pr-check-security.test.sh 158907 tests/fm-prepush-guard.test.sh 19208 tests/fm-procevent-when.test.sh 24358 tests/fm-procevent.test.sh 61916 tests/fm-public-followup.test.sh 35559 tests/fm-quota-array-dispatch-live-e2e.test.sh 12 -tests/fm-receipt-check.test.sh 96584 +tests/fm-receipt-check.test.sh 11636 tests/fm-receipt.test.sh 1331 tests/fm-reflect-skill.test.sh 372 tests/fm-remote-backlog-handoff.test.sh 15173 diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index e080574e44c..886f80fc175 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -162,7 +162,7 @@ case "$REMOTE_TIMEOUT" in *) [ "$REMOTE_TIMEOUT" -le 15 ] || REMOTE_TIMEOUT=5 ;; esac WATCHER_DOWNTIME_MARKER="$STATE/.watcher-down" -VALIDATION_PLAN_LOCK_STALE_SECS=30 +PR_PUBLICATION_LOCK_STALE_SECS=30 # The singleton-lock acquisition, EXIT trap, and the blocking supervision loop # all live below the source guard at the very bottom of this file (see "Main # entry"). Sourcing this file for unit tests therefore loads the functions - @@ -1764,14 +1764,14 @@ while :; do run_check_capture "$SCRIPT_DIR/fm-pr-poll.sh" --validated \ "$provider" "$url" "$host" "$path" "$number" || exit 1 out=$FM_CHECK_RESULT - elif [ -d "$STATE/.$id.validation-plan.lock" ] \ - && [ ! -L "$STATE/.$id.validation-plan.lock" ] \ - && validation_lock_age=$(fm_path_age "$STATE/.$id.validation-plan.lock") \ - && [ "$validation_lock_age" -ge 0 ] \ - && [ "$validation_lock_age" -lt "$VALIDATION_PLAN_LOCK_STALE_SECS" ] \ + elif [ -d "$STATE/.$id.pr-publication.lock" ] \ + && [ ! -L "$STATE/.$id.pr-publication.lock" ] \ + && publication_lock_age=$(fm_path_age "$STATE/.$id.pr-publication.lock") \ + && [ "$publication_lock_age" -ge 0 ] \ + && [ "$publication_lock_age" -lt "$PR_PUBLICATION_LOCK_STALE_SECS" ] \ && fm_pr_poll_artifacts_valid "$STATE" "$id" "$SCRIPT_DIR/fm-pr-poll.sh" defer-metadata; then # fm-pr-check publishes the authenticated poll tuple before its - # atomic metadata replacement while holding this transaction lock. + # atomic metadata replacement while holding its publication lock. # Retry on the next cycle instead of surfacing that brief, valid # pre-metadata state as an unauthenticated check. continue diff --git a/docs/architecture.md b/docs/architecture.md index e098c926b1c..a0b9f29ba16 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -10,7 +10,7 @@ firstmate's always-loaded operating contract and routing index for conditional p A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals whose crew is not provably working, authenticated check output such as PR merge polling or an X-mode mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS` without their own task worktree being written, idle panes whose board row stays in flight with no status line within `FM_IDLE_OPEN_WORK_SECS` of the last turn end or spawn record (`idle-with-open-work`) unless always-on liveness proves the crew is still working, declared external waits that remain paused past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. -Generated briefs bind each crewmate to report a terminal or `paused:` status within that bound while its board row remains in flight, and a `done:` claim is accepted only with its delivery artifact - a canonical PR URL for the PR modes, `ready in branch` for local-only, or the report path for a scout - with ship completion enforced by `bin/fm-receipt-check.sh --complete` and final cleanup enforced by `bin/fm-teardown.sh`. +Generated briefs bind each crewmate to report a terminal or `paused:` status within that bound while its board row remains in flight, and a `done:` claim is accepted only with its delivery artifact - a canonical PR URL for the PR modes, `ready in branch` for local-only, or the report path for a scout - with ship done acceptance enforced by `bin/fm-crew-state.sh`'s delivery gate (complete acceptance evidence from `bin/fm-receipt-check.sh` plus the PR registered by `bin/fm-pr-check.sh` or a clean ready branch) and final cleanup enforced by `bin/fm-teardown.sh`. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. Pane quietness plus the run step can miss a crew writing source, tests, or documentation behind a static pane, so a file newer than the start of the current quiet window in that task's recorded worktree defers the escalation. That deferral re-surfaces on the same `FM_PAUSE_RESURFACE_SECS` cadence as a declared wait, with a reason naming the write evidence rather than a wedge, and it is bounded to one pruned, depth-bounded, wall-clock-bounded walk (`FM_WORKTREE_WRITE_PRUNE`, `FM_WORKTREE_WRITE_MAXDEPTH`, `FM_WORKTREE_WRITE_TIMEOUT`) taken only in the branch that was about to escalate, never on every poll. diff --git a/docs/configuration.md b/docs/configuration.md index b9eee9c1645..5ed91af91d8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -508,7 +508,7 @@ Bootstrap reports missing GitHub authentication separately as `NEEDS_GH_AUTH`; [ [`bin/fm-bootstrap.sh`](../bin/fm-bootstrap.sh) owns the axi-family floor policy and the gh-axi and lavish-axi floors, while [`bin/fm-tasks-axi-lib.sh`](../bin/fm-tasks-axi-lib.sh) and [`bin/fm-quota-axi-lib.sh`](../bin/fm-quota-axi-lib.sh) hold their own tools' floor constants. This section is the single owner of that universal toolchain list; backend guides' prerequisites point here and add only their backend-specific tools. In that list, no-mistakes runs the validation pipeline, gh-axi, chrome-devtools-axi, and lavish-axi cover GitHub, browser, and rich-review operations, and tasks-axi plus quota-axi back backlog mutations and quota-aware array dispatch. -Full No-Mistakes PR registration and completion additionally require `python3` with its standard-library `sqlite3` module for the bound-run decision-evidence check owned by [`bin/fm-nm-run-lib.sh`](../bin/fm-nm-run-lib.sh). +Full No-Mistakes PR registration and completion additionally require `python3` with its standard-library `sqlite3` module for the run decision-evidence check owned by [`bin/fm-nm-run-lib.sh`](../bin/fm-nm-run-lib.sh). The per-backend delta is required only for the backend resolved from `FM_BACKEND`, then `config/backend`, then runtime auto-detection, then default `tmux`, so a home is never told to install a tool an inactive backend or feature would need. That delta is owned in code by `fm_backend_required_tools` in `bin/fm-backend.sh`: the resolved backend's own session-provider CLI (`tmux`, `herdr`, `zellij`, `orca`, or `cmux`), `jq` for the JSON-emitting experimental adapters (`herdr`, `zellij`, `cmux`) whose spawn and liveness paths parse the backend's JSON output, and the `treehouse` worktree provider for every session-provider-only backend (`tmux`, `herdr`, `zellij`, `cmux`). Backend tool availability uses the adapter's own executable resolver, so bootstrap and spawn agree on supported non-`PATH` locations such as cmux's bundled CLI. @@ -932,6 +932,7 @@ FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per no-mistakes query inside fm- FM_TODO_ITEM_MAX=100 # characters per projected session-todo item in bin/fm-todo-project.sh --emit FM_TODO_PR_TIMEOUT=20 # seconds allowed per direct forge poll in fm-todo-project; invalid or non-positive values reset to 20 FM_TEARDOWN_NM_TIMEOUT=10 # seconds allowed per no-mistakes query or abort inside fm-teardown.sh +FM_PR_CHECK_NM_TIMEOUT=10 # seconds allowed per no-mistakes status, CI-log, or decision-audit read inside fm-pr-check.sh FM_CREW_STATE_RUNS_LIMIT=200 # recent no-mistakes run rows scanned when axi status cannot be attributed directly FM_CREW_STATE_BIN=bin/fm-crew-state.sh # test override for the current-state reader used by working/paused watcher triage FMX_PAIRING_TOKEN= # X mode pairing token; .env opt-in authorizes replies and eligible lifecycle actions diff --git a/docs/gitlab-merge-watch.md b/docs/gitlab-merge-watch.md index 864b8f4ab25..a1c96ec481f 100644 --- a/docs/gitlab-merge-watch.md +++ b/docs/gitlab-merge-watch.md @@ -176,3 +176,4 @@ It refuses a GitLab merge request URL rather than sending it to the wrong forge, A GitLab task records no `pr_head=`. `gh` exposes the head commit as a selectable field, while plain `glab` exposes it only inside its JSON output, which would need a JSON processor firstmate does not require. Both consumers already treat it as optional: `bin/fm-teardown.sh` reads the head from the forge at teardown rather than from metadata and falls back to its provider-agnostic content check, and `bin/fm-review-diff.sh` resolves the head from the remote when none is recorded. +The one consumer that requires it is the no-mistakes PR-ready check in `bin/fm-pr-check.sh`, which compares the forge head with the pipeline run's `head_sha`; that is consistent with No-Mistakes publishing GitHub pull requests only, so a GitLab merge request is registered by `direct-PR` tasks alone. diff --git a/docs/scripts.md b/docs/scripts.md index e2d995dc3f4..3f24803198b 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -79,7 +79,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-local-default.sh` | Resolve the local default branch shared by readiness and guarded landing | | `fm-review-diff.sh` | Review a crewmate branch or resolved PR head against the authoritative base | | `fm-receipt.sh` | Append one validated acceptance-criterion evidence receipt to a ship task | -| `fm-receipt-check.sh` | Check ship evidence and own risk-based validation planning and completion | +| `fm-receipt-check.sh` | Check whether a ship task's declared acceptance criteria are accounted for | | `fm-receipt-schema.sh` | Validate the single receipt JSON schema used by append and read paths | | `fm-receipt-store.sh` | Own pinned ship contracts, evidence, metadata updates, and promotion storage | | `fm-marker-lib.sh` | Compatibility entry point for the from-firstmate carrier owned by `fm-operational-input.sh` | @@ -102,7 +102,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-supervisor-target-lib.sh` | Resolve the shared supervisor target and backend for the daemon and launcher | | `fm-supervise-daemon.sh` | Presence-gated away-mode sub-supervisor: self-handle routine wakes, guard injection by the detected primary harness, escalate batched digests, alert on failed delivery | | `fm-crew-state.sh` | Print one deterministic current-state line for a crew | -| `fm-nm-run-lib.sh` | Own shared no-mistakes run attribution and bound-run gate decision-evidence checks | +| `fm-nm-run-lib.sh` | Own shared no-mistakes run attribution and gate decision-evidence checks | | `fm-tangle-lib.sh` | Shared default-branch resolution and primary-checkout tangle classification | | `fm-supervision-lib.sh` | Shared in-flight-work-without-fresh-watcher-beacon predicate | | `fm-ff-lib.sh` | Shared guarded fast-forward helper for origin pulls and local secondmate syncs | diff --git a/docs/verification/evidence-receipts.md b/docs/verification/evidence-receipts.md index c2e260dc053..f6512757fcc 100644 --- a/docs/verification/evidence-receipts.md +++ b/docs/verification/evidence-receipts.md @@ -1,237 +1,105 @@ -# Evidence receipts and risk routing verification +# Evidence receipts verification -This record captures the active maintainer evidence for ship-task acceptance receipts and conservative validation routing as of 2026-10-04. -The exact receipt key and type schema is owned by the header and `--help` output of `bin/fm-receipt-schema.sh`; the criterion parser, classifier thresholds, metadata fields, and lifecycle commands are owned by the headers and help output of `bin/fm-receipt-check.sh`, `bin/fm-receipt.sh`, and `bin/fm-receipt-store.sh` at their respective executable boundaries. +This record captures the active maintainer evidence for ship-task acceptance receipts as of 2026-10-06. +Evidence receipts establish whether the implementing worker accounted for every acceptance criterion the ship brief declares; they certify nothing about review, CI, No-Mistakes completion, or merge readiness. +The exact receipt key and type schema is owned by the header and `--help` output of `bin/fm-receipt-schema.sh`; the criterion parser and accounting contract are owned by `bin/fm-receipt-check.sh`, the writer by `bin/fm-receipt.sh`, and pinned storage by `bin/fm-receipt-store.sh`, each at its executable boundary. +Delivery gates that consume the accounting result are owned by `bin/fm-pr-check.sh` (PR-ready) and `bin/fm-crew-state.sh` (done acceptance); No-Mistakes run attribution and the ask-user decision audit are owned by `bin/fm-nm-run-lib.sh`. ## Guarantees under test - New ship briefs receive stable acceptance-criterion ids plus an empty append-only evidence ledger and lock through one atomic pinned-directory publication, while scout and secondmate scaffolds remain outside the receipt contract. - Concurrent ship scaffolds use one exclusive brief identity, so a losing invocation cannot remove the winning brief or evidence contract. - Ship scaffold output requires replacing both task and acceptance-criterion placeholders, and spawn refuses unresolved task text or criteria before endpoint creation. -- Legacy ship briefs without an acceptance-criteria section may still launch with an explicit migration warning, but the completion gate remains parked until Firstmate installs a valid evidence contract. -- Every ship completion, including a promoted scout, remains parked until a valid acceptance contract and structurally valid receipts account for every required criterion. -- Low-risk routing is limited to CHANGELOG formatting or reflow changes whose non-whitespace byte sequence is identical and which have file-bound mechanical proof, while every content-byte or uncertain change defaults high. -- High-risk, broad, sensitive, weakly proven, materially expanded, or uncertain changes retain full No-Mistakes validation. -- `direct-PR` and `local-only` retain the evidence gate and current-state reconciliation without entering No-Mistakes. -- The explicit implementation-complete action records one current timestamp for the current clean commit, refreshes that timestamp when the head changes, remains idempotent for the same head, and supplies the plan interval origin. -- Completion requires observed post-plan mechanical evidence, the exact No-Mistakes run created with the latest unguessable plan generation and bound to its path and head with current checks-green status or CI-log evidence plus the decision-evidence check owned by `bin/fm-nm-run-lib.sh`, a GitHub PR with forge-observed exact-head metadata, a supported non-GitHub direct-PR with the existing canonical HTTPS PR URL predicate and no head observation, or a clean fast-forward-ready branch. +- Legacy ship briefs without an acceptance-criteria section may still launch with an explicit migration warning, but done acceptance remains parked until Firstmate installs a valid evidence contract. +- `fm-receipt-check.sh ` offers only the default check, `--criterion`, and `--parse-criteria`, emits one `fm-evidence-check.v2` object with `required`, `evidenced`, `accepted_blocked`, `missing`, and `invalid`, refuses every removed validation action as an unknown option, and writes no task metadata. +- The latest structurally valid receipt per criterion decides it: only `outcome=success` evidences a criterion, a later failure receipt revokes an earlier success until a fresh success lands, and `result` stays descriptive so an expected observation such as `401` recorded as success is evidence. +- A receipt naming an undeclared criterion, a malformed record, or a blank line makes the ledger invalid rather than silently disappearing. +- `outcome=accepted-blocked` is valid only with a non-empty `captain_exception` reference recorded verbatim; such a criterion is accounted for without being evidenced and is reported in the distinct always-present `accepted_blocked` list, never in `evidenced`. +- The checker never consults No-Mistakes: a tripwire `no-mistakes` binary records zero invocations across the whole accounting suite. - Receipt append and check share one executable owner that resolves and pins every raw data-path component inside the store process, opens and verifies the task directory relative to that pinned parent, and then opens relative no-follow brief and single-link ledger paths portably on Linux and macOS. - Receipt storage physicalizes the trusted Firstmate-home prefix for standard system symlinks, then retains no-follow checks for the data suffix, task directory, and task artifacts. -- Promotion pins and verifies its scout task directory before reading or replacing the brief and ledger, and refuses symlinked or out-of-root task paths before mutation. -- Promotion retries distinguish identity-bound unfinished rollback from committed retirement recovery, while post-commit reporting cannot reverse durable success. -- Planning retains the pinned shared ledger lock through metadata publication, so only receipts appended after the published plan boundary qualify as fresh mechanical evidence. -- The checker parent owns a read/write release descriptor before snapshot spawn, so early child failures cannot block cleanup waiting for a FIFO reader. -- Snapshot readiness status is checked and terminal on publication failure; hold documents ready `0`, refusal `1`, missing-ledger `3`, and pinned non-ship `4` statuses. -- Receipt append, check, and promotion consume one executable acceptance-criterion parser that requires nonblank descriptions. -- Structurally valid receipts require non-whitespace summary and result strings plus an explicit structured outcome; only `outcome=success` evidences a criterion, while failure, negative, zero, skipped, empty, placeholder, weak, and legacy outcomes remain unevidenced and `result` stays descriptive so expected observations such as `401` are unambiguous. -- `outcome=accepted-blocked` is valid only with a non-empty `captain_exception` reference recorded verbatim (the date plus the captain's own words or the board key that holds them); a criterion whose latest receipt is a valid accepted-blocked is accounted for without being evidenced, so planning, readiness, and completion can proceed while the evidence check reports it in the distinct always-present `accepted_blocked` list, never in `evidenced`. -- A task carrying any accepted-blocked criterion is never auto-merged; plan, readiness, and completion output surfaces the criteria and their exception references plainly so the PR description states them. -- Head-bound receipts store only the exact canonical 40- or 64-character lowercase hexadecimal commit id reported by Git. -- The ship brief's acceptance-evidence template instructs workers to copy any cited worktree-resident artifact into `data//artifacts/` before `done:` and cite the copied path, and `bin/fm-receipt.sh` warns on a relative `--artifact` path without refusing the append. - Receipt append holds a stable task lock, copies the canonical single-link ledger plus one complete record to a synced mode-0600 single-link temporary file, and atomically renames it over the canonical ledger so concurrent hard-link aliases retain the old inode. -- Criterion parsing rejects known scaffold placeholder tokens in balanced or unmatched brace forms while allowing concrete brace syntax such as JSON examples. -- One shared cleanliness predicate requires `git status` with submodule ignores disabled to succeed with empty tracked, staged, untracked, and submodule output for implementation completion, planning, binding, terminal completion, and final done acceptance. -- Every plan requires a resolved authoritative base and commit diff before it can publish any delivery path. -- Every diff input, including the special-mode summary probe, must execute successfully before risk classification. -- Normal and promoted ship briefs consume the same executable acceptance-evidence and per-mode delivery renderer. -- The pinned brief and task metadata must record the same concrete delivery mode before validation can proceed. -- Ship state requires exactly one valid recorded delivery mode before any No-Mistakes lookup. -- Findings that invalidate a receipt or acceptance claim atomically bind one generation-scoped idempotent finding-to-criterion marker to the invalidation-time head and receipt boundary, then require a strict non-empty descendant delta and a later successful receipt bound to the new head before replanning or completion. -- One pinned state-directory owner snapshots single-link no-follow metadata and performs compare-bound atomic replacements for every validation metadata update. -- PR registration publishes canonical PR identity and its validation publication generation through one compare-bound pinned metadata replacement after the watcher artifacts publish, and revokes those artifacts if that replacement fails. -- Binding fixtures cover planned heads and faithful restamps separately from advances and preplan recovery, using the eligibility rules owned by `bin/fm-receipt-check.sh`. -- Recovery fixtures exercise late planning, mid-run rebasing, and identical-replan refusal through the binding and completion contract owned by `bin/fm-receipt-check.sh`. -- No-Mistakes status, intent, and CI-log observations use the shared bounded call boundary. -- Every completion requires path-specific terminal evidence and records its plan path and authoritative completed head. -- Head-accounting fixtures cover descendants, faithful restamps, descendants of restamps, and owned content identity, with refusal cases for unvalidated checkout content, mismatched branches, incomplete ownership evidence, failed runs, and cancelled runs. -- Local-only readiness and guarded landing consume one fail-closed executable default-branch resolver. -- Planning and completion refuse tracked, staged, or untracked worktree changes. -- Initial planning accepts a caller base only when it equals the repository's authoritative merge boundary, so a later ancestor cannot hide earlier task commits. -- Ordinary No-Mistakes findings return to the original worker through guarded custody return and then full revalidation. -- Direct-PR registration publishes its watcher before recording completion, while other paths preserve their earlier path-specific completion boundary. - -## Head-accounting regression coverage - -The binding and completion contract is owned by the header and help of [`bin/fm-receipt-check.sh`](../../bin/fm-receipt-check.sh); the tree, chain-provenance, and branch-ownership predicates are owned by [`bin/fm-nm-run-lib.sh`](../../bin/fm-nm-run-lib.sh). -The executable-interface fixtures in [`tests/fm-receipt-check.test.sh`](../../tests/fm-receipt-check.test.sh) exercise these guarantees: - -- The late-plan fixture records an already-existing run as the plan boundary, refuses it without ownership, then checks, binds, and completes its terminal pass through content identity without changing the plan. -- The mid-run rebase fixture advances main with real content, replays the task chain, and adds pipeline fixes, asserting that the resulting head neither descends from the planned head nor shares its tree before checking binding and completion. -- The replan fixture refuses an identical plan without clearing the live binding, then verifies that changed-content replanning clears stale run and completion bindings. -- The synchronized fixture checks completion with a self-submitted descendant after content binding and after ancestry binding transitions from pipeline custody to synchronized ownership. -- The rebase fixture also checks fleet done acceptance through `bin/fm-crew-state.sh` with original and refreshed implementation heads, rejecting stale completion generations and mismatched paths. -- The refusal fixtures separate tree identity from ownership and branch identity, and reject run or checkout content that the run did not validate, as well as failed and cancelled runs. -- The binding-check prerequisite fixture verifies the single read-only verdict surface for missing or invalid receipts, dirty or missing worktrees, invalid plans or generations, and unobservable runs. - -Commands and captured results are recorded below. +- The writer stamps no commit head onto a receipt; the schema still reads legacy head-stamped records. +- Promotion pins and verifies its scout task directory before reading or replacing the brief and ledger, refuses symlinked or out-of-root task paths before mutation, and distinguishes identity-bound unfinished rollback from committed retirement recovery. +- Receipt append, check, and promotion consume one executable acceptance-criterion parser that requires nonblank descriptions and rejects scaffold placeholder tokens while allowing concrete brace syntax. +- The pinned brief and task metadata must record the same concrete delivery mode before accounting proceeds. +- Initial ship PR registration exercises the acceptance-evidence gate owned by `bin/fm-pr-check.sh` and names missing or invalid criteria when it refuses; direct-PR registration never consults No-Mistakes. +- Initial no-mistakes PR registration exercises the run-identity gate owned by `bin/fm-pr-check.sh`, records `nm_run_id=`, and refuses runs on another branch or PR, foreign heads, failed, cancelled, unfinished, or unobservable runs. +- The PR-ready and done-acceptance audit fixtures exercise the decision-evidence predicate owned by `bin/fm-nm-run-lib.sh`, including refusal of unmatched answers and unreadable run data; the conditional done-time audit is owned by `bin/fm-crew-state.sh`. +- `bin/fm-crew-state.sh` accepts a ship done only with a clean worktree, complete evidence, and `pr=` recorded for the PR modes or a clean checked-out `fm/` branch for local-only. +- `bin/fm-spawn.sh --relaunch` carries `pr=`, `pr_head=`, and `nm_run_id=` into the replacement record, including a restart mid-handoff, and invents none for an unregistered task. +- PR registration publishes canonical PR identity through one compare-bound pinned metadata replacement after the watcher artifacts publish, revokes those artifacts if that replacement fails, serializes per task on `state/..pr-publication.lock`, and the watcher defers a valid pre-metadata poll only while that lock is fresh. ## Known limitations -- Claim invalidation reads the worktree head immediately before acquiring the validation metadata lock, so an unsupported concurrent ref mutation can bind the marker to the earlier head; the single-operator workflow excludes that mutation during validation. -- PR registration snapshots and replaces metadata through separate pinned-store processes, so an unsupported concurrent byte-identical state-directory swap can move the transaction to the replacement directory; the single-operator workflow excludes state-directory replacement during validation. +- PR registration snapshots and replaces metadata through separate pinned-store processes, so an unsupported concurrent byte-identical state-directory swap can move the transaction to the replacement directory; the single-operator workflow excludes state-directory replacement during registration. +- The ask-user decision audit compares recorded gate responses with status-ledger records; it proves matching process records, not authenticated authorship (`bin/fm-classify-lib.sh` owns that limitation). ## Verification environment -- Date: 2026-10-04. +- Date: 2026-10-06. - ShellCheck: 0.11.0. - Git: 2.34.1. ## Commands and results -On 2026-10-04, the focused completion regressions and both owning suites passed with the command below (exit 0). -The synchronized fixture exercises completion with a self-submitted descendant head after content binding and after ancestry binding transitions from pipeline custody to synchronized ownership. -The rebase fixture exercises fleet done acceptance with both original and refreshed implementation heads and refuses stale completion generations and mismatched paths. +The owning suites passed with these exact commands on 2026-10-06 (each exit 0). ```text -$ TMPDIR="$PWD/.review-tmp" bash -c 'bash tests/fm-receipt-check.test.sh && bash tests/fm-crew-state.test.sh' -ok - converged synchronized binding requires the full run-owned sync evidence -ok - mid-run rebase onto newer main binds and completes by content identity -all fm-crew-state tests passed -``` - -The ownership-transition and prerequisite-refusal regressions were refreshed on 2026-10-04 with `TMPDIR="$PWD/.review-tmp" bash tests/fm-receipt-check.test.sh` (exit 0), including `ok - converged synchronized binding requires the full run-owned sync evidence` and `ok - bind-check prerequisite refusals preserve one read-only binding verdict contract`. - -The focused behavioral suites passed with these exact commands. - -```text -$ tests/fm-receipt.test.sh -ok - fm-receipt appends one compact validated receipt -ok - fm-receipt preserves prior records and accepts --result -ok - fm-receipt warns on relative --artifact while still appending -ok - fm-receipt gates accepted-blocked on a verbatim captain exception -ok - fm-receipt stores and validates an exact canonical commit id -ok - fm-receipt appends complete large JSONL records -ok - fm-receipt rejects invalid types, ids, missing results, and undeclared criteria -ok - fm-receipt uses portable paths and rolls back incomplete appends -ok - fm-receipt refuses non-ship tasks and unsafe ledger paths -ok - fm-receipt rejects task-directory replacement before its no-follow open -ok - fm-receipt rejects data-directory replacement before its pinned open -ok - fm-receipt rejects regular data replacement after pinning -ok - fm-receipt atomically replaces the ledger without mutating hard-link aliases -ok - fm-receipt physicalizes trusted home prefixes but rejects data symlinks - -$ tests/fm-receipt-check.test.sh -ok - fm-receipt-check help renders an executable generation-bound bind command -ok - fm-receipt-check reports required, evidenced, and missing ids deterministically -ok - fm-receipt-check distinguishes complete evidence from invalid JSONL -ok - structured success and negative outcomes control criterion evidence +$ bash tests/fm-receipt-check.test.sh +ok - complete evidence reports the fm-evidence-check.v2 shape and exits 0 +ok - a missing criterion is named and exits 1 +ok - failure never satisfies, expected-negative success does, and the latest receipt per criterion wins +ok - unknown criteria and malformed records make the ledger invalid instead of vanishing +ok - accepted-blocked accounts for its criterion visibly without evidencing it +ok - receipt append, --criterion, and --parse-criteria consume one criterion grammar +ok - fm-receipt-check offers only accounting actions and writes no validation metadata ok - pinned brief and metadata delivery modes must match exactly -ok - pinned metadata owner rejects hard-linked validation records +ok - pinned metadata owner rejects hard-linked task records ok - invalid ship briefs fail and scout/report behavior stays unchanged ok - early snapshot failures release cleanup without a FIFO reader ok - snapshot readiness publication failures terminate without waiting ok - fm-receipt-check pins task evidence and rejects hard-linked ledgers -ok - receipt append and check consume one criterion grammar -ok - exact bound runs complete from the shared current CI-log readiness predicate -ok - finding-to-criterion invalidations remain inspectable in task metadata -ok - run binding resolves abbreviated heads and rejects non-planned commits -ok - binding and completion work against the real agent-supplied intent-log shape while wrong runs fail closed -ok - terminal passed runs seal their own pipeline advance and refuse foreign drift -ok - pipeline rebase restamps bind and seal their validated content -ok - restamped chains enforce provenance and ownership -ok - active pipeline-owned descendant binds without replan -ok - terminal pipeline-owned descendant binds and completes -ok - restamp chain followed by pipeline doc commit binds and completes -ok - active descendant binds using axi sync fallback when axi status omits branch_sync -ok - unowned active descendant binding is rejected -ok - converged synchronized run binds and completes while monitoring its PR -ok - converged synchronized binding requires the full run-owned sync evidence -ok - descendant bind rejects the wrong branch -ok - low-risk mechanical changes can skip a full No-Mistakes run -ok - low risk requires safe changelog prose and file-bound mechanical evidence -ok - implementation completion refreshes per head and remains idempotent -ok - plan publication holds the pinned ledger boundary against concurrent receipts -ok - diff summary errors fail closed before risk classification -ok - successful terminal runs bind while failed runs remain rejected -ok - No-Mistakes status and CI-log observations are bounded -ok - authoritative documentation remains high -ok - terminal delivery paths record one completion timestamp at their boundary -ok - completion signals release the validation lock for retry -ok - replanning invalidates prior run and completion bindings -ok - a passed run recorded before its plan binds and completes by content identity -ok - mid-run rebase onto newer main binds and completes by content identity -ok - content-identity binding still refuses unreviewed content and unowned or failed runs -ok - dirty worktrees cannot be planned or completed -ok - git status errors fail implementation, planning, and completion cleanliness gates -ok - shared cleanliness inspects ignored submodules -ok - direct and local plans never invoke No-Mistakes -ok - local completion requires fast-forward readiness -ok - local readiness and landing share one fail-closed default resolver -ok - security and uncertain changes retain full No-Mistakes validation -ok - direct-PR and local-only retain evidence gates without invoking No-Mistakes -ok - completion refuses a standing done: claim that carries no delivery artifact -ok - accepted-blocked accounts for its criterion without evidencing it and still refuses real gaps - -$ tests/fm-crew-state.test.sh -ok - ship completion requires evidence and current-head implementation completion +ok - receipt accounting never consults No-Mistakes + +$ bash tests/fm-pr-check-handoff.test.sh +ok - direct-PR handoff proceeds on complete evidence and never consults No-Mistakes +ok - no-mistakes handoff records nm_run_id from a passed run matching branch, PR, and head +ok - an active run whose CI log reads green is PR-ready +ok - runs on another branch or PR, foreign heads, failed, unfinished, or unobservable runs never arm +ok - PR-ready refuses a self-answered ask-user finding until a firstmate decision record exists +ok - unreadable No-Mistakes decision data refuses PR-ready with its own reason + +$ bash tests/fm-crew-state.test.sh +ok - ship completion requires complete acceptance evidence ok - ship completion fails closed when the evidence contract is malformed -ok - run-step done requires current-generation validation completion -ok - status-log done requires existing plan completion -ok - final done requires a clean inspectable worktree -ok - LOW validation remains parked until PR completion -ok - direct-PR and local-only state reads skip No-Mistakes -ok - ship state requires one valid mode before run lookup +ok - PR-mode done requires the PR registered by fm-pr-check and a clean worktree +ok - local-only done requires the clean fm/ branch and no PR +ok - done acceptance applies the ask-user decision audit to the recorded run all fm-crew-state tests passed -$ tests/fm-brief.test.sh -ok - fm-brief.sh: no-mistakes/direct-PR/local-only briefs generate cleanly -ok - fm-brief: scout and secondmate code paths still scaffold well-formed briefs -ok - fm-brief: concurrent ship scaffolds preserve one complete owner -ok - fm-brief: ship evidence publication is atomic and retryable +$ env -u FM_TASK_ID bash tests/fm-spawn-relaunch-dead-endpoint.test.sh +ok - fm-spawn --relaunch: pr=, pr_head=, and nm_run_id= survive a restart mid-handoff +ok - fm-spawn --relaunch: a task not yet registered gains no empty delivery records -$ bin/fm-lint.sh -fm-lint.sh: ShellCheck 0.11.0 (pinned 0.11.0) -exit 0 +$ env -u FM_TASK_ID bash tests/fm-pr-check-security.test.sh +ok - PR registration serializes on the per-task publication lock and releases it +ok - watcher defers valid pre-metadata polls while the publication lock is held +ok - watcher bounds pre-metadata deferral by publication lock freshness + +$ bash tests/fm-receipt.test.sh +ok - fm-receipt writes no commit head while the schema still reads legacy head-stamped records +ok - fm-receipt gates accepted-blocked on a verbatim captain exception ``` -The named safety regressions also passed. +`bash tests/fm-brief.test.sh` and `bash tests/fm-task-delivery.test.sh` passed on the same date, asserting that no generated or promoted ship brief instructs a removed receipt-check action or carries a plan generation. +`bin/fm-lint.sh` exited 0 with ShellCheck 0.11.0. -```text -$ tests/fm-watch-triage.test.sh -exit 0 - -$ bash tests/fm-ask-user-authority.test.sh -ok - primary workers and secondmates receive the authority rule through generated instructions - -$ tests/fm-tangle-guard.test.sh -ok - fm-brief: ship brief asserts worktree isolation before the branch step -ok - fm-spawn: aborts unless the resolved worktree is a genuine, isolated worktree - -$ tests/fm-pr-merge.test.sh -ok - fm-pr-merge records pr= and pr_head= before invoking gh-axi pr merge -ok - fm-pr-merge refuses before merging when task meta is missing - -$ tests/fm-pr-check-security.test.sh -ok - valid direct and merge flows record exact metadata and reject multiline head metadata -ok - PR registration serializes with validation planning -ok - fast PR registration completes and keeps its watcher armed -ok - PR metadata publication rejects post-snapshot redirection -exit 0 - -$ tests/fm-task-delivery.test.sh -ok - fm-spawn: a ship spawn requires a valid explicit mode and yolo before anything is created -ok - fm-promote-transaction: help renders successfully -ok - fm-spawn: unresolved task and criterion placeholders refuse before launch -ok - fm-spawn: legacy ship briefs launch but disclose deferred evidence migration -ok - fm-spawn: scout and secondmate spawns refuse ship delivery flags -ok - fm-spawn: the brief's recorded mode and the spawn's explicit mode must agree -ok - fm-spawn: a rigor downgrade against the registered posture is announced, never blocked -ok - fm-spawn: a scout spawn resolves no delivery posture from the registry -ok - fm-promote: promotion installs a fail-closed ship evidence contract -ok - fm-promote: symlinked task directories refuse before mutation -ok - fm-promote: configured data symlinks remain visible to no-follow pinning -ok - fm-promote: concurrent losers cannot remove the winner lock -ok - fm-promote: signal-terminated transactions fail closed -ok - fm-promote: interrupted task replacement rolls back atomically -ok - fm-promote: store signals before commit roll back both replacements -ok - fm-promote: intermediate state symlinks fail closed -ok - fm-promote: state path replacement cannot redirect metadata -ok - fm-promote: retry recovers an identity-bound crashed transaction -ok - fm-promote: post-commit reporting cannot reverse success -ok - fm-promote: committed retirement recovery preserves ship state -ok - fm-project-mode: the conditional policy is accepted, mapped for mechanical callers, and readable raw -# all fm-task-delivery tests passed - -$ tests/fm-teardown-endpoint-safety.test.sh -ok - fm-teardown: missing, empty, malformed, ambiguous, and task-mismatched endpoints refuse before every mutation or runtime call -``` +## Line accounting + +`git diff --numstat dad3e4a58cac6f3f450523b9dcd72e754d8e1753 44dbd3f1483e4050f93cff9e85d6436f870d7f67 -- bin/ tests/` reports the following totals for the reviewed change. + +| Scope | Lines removed | Lines added | +| --- | ---: | ---: | +| Production (`bin/`) | 1,550 | 297 | +| Tests (`tests/`) | 2,892 | 750 | diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 3332b0eda73..cdfb6df3b84 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -231,19 +231,26 @@ test_ship_modes_generate_clean_briefs() { assert_grep "{TASK}" "$brief" "$id: brief missing the {TASK} placeholder" assert_grep "mid-task \`working:\` line (including setup complete) is nonterminal" "$brief" \ "$id: brief missing nonterminal working:/setup-complete gate protection" + assert_grep "certify nothing about review, CI, No-Mistakes completion, or merge readiness" "$brief" \ + "$id: brief did not state the accounting-only scope of receipts" + assert_grep "record a \`--outcome failure\` receipt for it, fix, and record a fresh \`--outcome success\`" "$brief" \ + "$id: brief omitted the receipt invalidation rule" + case "$(grep -o 'fm-receipt-check.sh [^ ]* --[a-z-]*' "$brief" || true)" in + '') ;; + *) fail "$id: brief still instructs a removed receipt-check action: $(grep -o 'fm-receipt-check.sh [^ ]* --[a-z-]*' "$brief")" ;; + esac + assert_no_grep "Firstmate-Validation-Generation" "$brief" "$id: brief still carries the plan generation" case "$mode" in direct-PR) - assert_grep "fm-receipt-check.sh $id --plan" "$brief" "$id: direct-PR brief omitted validation start" - assert_grep "canonical PR-ready helper records the observed completion" "$brief" "$id: direct-PR brief omitted observed PR completion" + assert_grep "canonical PR-ready helper registers the PR" "$brief" "$id: direct-PR brief omitted PR registration" ;; local-only) - assert_grep "fm-receipt-check.sh $id --plan" "$brief" "$id: local-only brief omitted validation start" - assert_grep "fm-receipt-check.sh $id --complete" "$brief" "$id: local-only brief omitted branch-ready completion" + assert_grep "done: ready in branch fm/$id" "$brief" "$id: local-only brief omitted its ready report" ;; no-mistakes) - assert_grep "fm-receipt-check.sh $id --complete" "$brief" "$id: no-mistakes brief omitted pipeline completion" - assert_grep "fm-receipt-check.sh $id --bind-run" "$brief" "$id: no-mistakes brief omitted exact run binding" - assert_grep "ordinary findings from any validation tier" "$brief" "$id: no-mistakes brief narrowed original-worker fixes" + assert_grep "Firstmate will then start full No-Mistakes validation on this worker" "$brief" "$id: no-mistakes brief omitted the full-validation handoff" + assert_grep "fix ordinary findings only after Firstmate directs" "$brief" "$id: no-mistakes brief narrowed original-worker fixes" + assert_grep "branch, PR URL, full head SHA, and outcome" "$brief" "$id: no-mistakes brief omitted how PR-ready proves the run" ;; esac assert_no_grep "EOF" "$brief" "$id: brief leaked a heredoc EOF marker (unterminated heredoc)" @@ -299,8 +306,8 @@ test_ship_mode_is_explicit_not_registry() { brief="$home/data/brief-explicit-a5/brief.md" grep -qx "Delivery contract: mode=no-mistakes" "$brief" \ || fail "registered direct-PR posture overrode the explicit --mode" - assert_grep "Firstmate will then classify validation risk" "$brief" \ - "explicit no-mistakes brief did not render risk-based validation" + assert_grep "Firstmate will then start full No-Mistakes validation on this worker" "$brief" \ + "explicit no-mistakes brief did not render the full-validation handoff" # An unregistered project is not a blocker either, because nothing is looked up. FM_HOME="$home" "$ROOT/bin/fm-brief.sh" brief-explicit-a6 never-registered --mode local-only >/dev/null 2>&1 \ @@ -1033,7 +1040,7 @@ test_firstmate_repo_ship_brief_prefills_verification_criterion() { "firstmate-repo scaffold dropped the placeholder-replacement instruction" brief="$home/data/brief-fm-ac99-$mode/brief.md" # shellcheck disable=SC2016 # Literal backticks must remain unexpanded. - assert_grep '- AC99: changed tests green via `bin/fm-test-run.sh --changed` and `FM_LINT_JOBS=1 bin/fm-lint.sh` clean, recorded as an evidence line with the branch head before validation planning;' "$brief" \ + assert_grep '- AC99: changed tests green via `bin/fm-test-run.sh --changed` and `FM_LINT_JOBS=1 bin/fm-lint.sh` clean, recorded as an evidence line naming the branch head before the implementation-complete report;' "$brief" \ "firstmate-repo AC99 cannot be evidenced before PR creation" assert_grep 'broad regression is owned by the PR GitHub CI per .no-mistakes.yaml' "$brief" \ "firstmate-repo AC99 did not identify CI as the broad regression owner" diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 504f2d46563..e3abffef5fe 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -35,6 +35,40 @@ set -u CREW_STATE="$ROOT/bin/fm-crew-state.sh" TMP_ROOT=$(fm_test_tmproot fm-crew-state) fm_git_identity fmtest fmtest@example.invalid +command -v python3 >/dev/null 2>&1 || { echo "skip: python3 not found (done-time ask-user audit)"; exit 0; } + +# The done-time ask-user audit reads the no-mistakes state database read-only; +# the suite gets an isolated fixture with every fake run id it reports, so a +# fixture done never reads the operator's real daemon state. +NM_DIR="$TMP_ROOT/nm-home" +mkdir -p "$NM_DIR" +python3 - "$NM_DIR" <<'PY' +import os, sqlite3, sys +db = sqlite3.connect(os.path.join(sys.argv[1], "state.sqlite")) +db.executescript(""" +CREATE TABLE runs (id TEXT PRIMARY KEY); +CREATE TABLE step_results ( + id TEXT PRIMARY KEY, run_id TEXT, step_name TEXT, step_order INTEGER, status TEXT, + findings_json TEXT, approval_reason TEXT, override_reason TEXT, skip_reason TEXT +); +CREATE TABLE step_rounds ( + id TEXT PRIMARY KEY, step_result_id TEXT, round INTEGER, selection_source TEXT, + selected_finding_ids TEXT, findings_json TEXT, user_findings_json TEXT +); +INSERT INTO runs VALUES ('01RUN'), ('01RUNLIVE'); +""") +db.commit() +PY +export NM_HOME="$NM_DIR" + +nm_db() { # + python3 - "$NM_DIR" "$1" <<'PY' +import os, sqlite3, sys +db = sqlite3.connect(os.path.join(sys.argv[1], "state.sqlite")) +db.executescript(sys.argv[2]) +db.commit() +PY +} # A real git repo checked out on , so the helper's branch attribution # (git symbolic-ref) resolves like it would for a live crew worktree. @@ -141,7 +175,7 @@ make_no_timeout_toolbin() { # -> echoes toolbin path # Run the helper for one case dir. FM_FAKE_* env (run output, busy flag) are read # from the caller's environment by the fakes above. run_crew_state() { # - local case_dir=$1 id=$2 brief ledger fixture_head fixture_mode added_mode=0 + local case_dir=$1 id=$2 brief ledger fixture_mode if grep -qx 'kind=ship' "$case_dir/state/$id.meta" 2>/dev/null; then fixture_mode=$(sed -n 's/^mode=//p' "$case_dir/state/$id.meta" | tail -1) [ -n "$fixture_mode" ] || fixture_mode=no-mistakes @@ -162,16 +196,10 @@ EOF printf '%s\n' '{"criterion":"AC1","type":"review","outcome":"success","summary":"fixture evidence","result":"complete"}' > "$ledger" fi [ -e "$case_dir/data/$id/.evidence.lock" ] || : > "$case_dir/data/$id/.evidence.lock" + # A fixture that never chose a mode is a registered no-mistakes task whose + # PR fm-pr-check already recorded, so run-step done reads as done. if ! grep -q '^mode=' "$case_dir/state/$id.meta" 2>/dev/null; then - printf 'mode=no-mistakes\n' >> "$case_dir/state/$id.meta" - added_mode=1 - fi - if [ "$added_mode" -eq 1 ]; then - fixture_head=$(git -C "$case_dir/wt" rev-parse HEAD 2>/dev/null || true) - if [ -n "$fixture_head" ]; then - printf 'implementation_completed_at=1\nimplementation_completed_head=%s\nvalidation_generation=legacy-fixture\nvalidation_path=full-no-mistakes\nvalidation_head=%s\nvalidation_completed_generation=legacy-fixture\nvalidation_completed_path=full-no-mistakes\nvalidation_completed_head=%s\n' \ - "$fixture_head" "$fixture_head" "$fixture_head" >> "$case_dir/state/$id.meta" - fi + printf 'mode=no-mistakes\npr=https://github.com/o/r/pull/1\n' >> "$case_dir/state/$id.meta" fi fi PATH="$case_dir/fakebin:$PATH" FM_STATE_OVERRIDE="$case_dir/state" FM_DATA_OVERRIDE="$case_dir/data" "$CREW_STATE" "$id" @@ -1515,7 +1543,7 @@ test_ship_done_is_held_until_evidence_is_complete() { mkdir -p "$d/data/$id" cat > "$d/data/$id/brief.md" <<'EOF' # Task -Exercise the completion evidence gate. +Exercise the evidence gate. # Acceptance criteria - AC1: The implementation works. @@ -1526,8 +1554,9 @@ Delivery contract: mode=no-mistakes EOF : > "$d/data/$id/evidence.jsonl" : > "$d/data/$id/.evidence.lock" - fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" "mode=no-mistakes" - printf 'done: implementation complete\n' > "$d/state/$id.status" + fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" "mode=no-mistakes" \ + "pr=https://github.com/o/r/pull/1" + printf 'done: PR https://github.com/o/r/pull/1 checks green\n' > "$d/state/$id.status" arm_idle_record "$d/state" "$id" out=$(PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" FM_DATA_OVERRIDE="$d/data" "$CREW_STATE" "$id") assert_contains "$out" "state: parked" "missing evidence must prevent done acceptance" @@ -1541,20 +1570,10 @@ EOF assert_contains "$out" "missing evidence: AC2" "failed outcome must leave its criterion missing" FM_DATA_OVERRIDE="$d/data" FM_STATE_OVERRIDE="$d/state" "$ROOT/bin/fm-receipt.sh" "$id" AC2 lint "regression checks" "passed" --outcome success >/dev/null out=$(PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" FM_DATA_OVERRIDE="$d/data" "$CREW_STATE" "$id") - assert_contains "$out" "state: parked" "complete evidence must still require implementation completion" - assert_contains "$out" "source: implementation-gate" "missing implementation completion must name its gate" - FM_DATA_OVERRIDE="$d/data" FM_STATE_OVERRIDE="$d/state" "$ROOT/bin/fm-receipt-check.sh" "$id" --implementation-complete >/dev/null \ - || fail "implementation completion fixture could not be recorded" - out=$(PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" FM_DATA_OVERRIDE="$d/data" "$CREW_STATE" "$id") - assert_contains "$out" "state: done" "current-head implementation completion must release done acceptance" + assert_contains "$out" "state: done" "complete evidence with a registered PR must release done acceptance" assert_contains "$out" "source: status-log" "released completion retains status-log source" - printf 'head change\n' >> "$d/wt/file.txt" - git -C "$d/wt" add file.txt - git -C "$d/wt" commit -q -m 'advance implementation head' - out=$(PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" FM_DATA_OVERRIDE="$d/data" "$CREW_STATE" "$id") - assert_contains "$out" "state: parked" "stale implementation completion escaped after a head change" - assert_contains "$out" "source: implementation-gate" "stale implementation completion must name its gate" - pass "ship completion requires evidence and current-head implementation completion" + grep -q 'validation_\|implementation_completed' "$d/state/$id.meta" && fail "done acceptance wrote validation metadata" + pass "ship completion requires complete acceptance evidence" } test_ship_done_with_malformed_brief_fails_closed() { @@ -1571,7 +1590,8 @@ Exercise a pre-evidence ship brief. # Definition of done Delivery contract: mode=direct-PR EOF - fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" "mode=direct-PR" + fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" "mode=direct-PR" \ + "pr=https://github.com/o/r/pull/10" printf 'done: PR https://github.com/o/r/pull/10\n' > "$d/state/$id.status" arm_idle_record "$d/state" "$id" out=$(PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" FM_DATA_OVERRIDE="$d/data" "$CREW_STATE" "$id") @@ -1581,99 +1601,98 @@ EOF pass "ship completion fails closed when the evidence contract is malformed" } -test_run_step_done_requires_current_plan_completion() { +# One handoff per delivery mode: done is accepted only once the mode's delivery +# record exists - pr= written by fm-pr-check for the PR modes, the clean fm/ +# branch for local-only - with no validation metadata anywhere. +test_pr_modes_done_requires_registered_pr() { reset_fakes - local d out id=validation-stage head - d=$(new_case validation-stage) - make_repo_on_branch "$d/wt" "fm/$id" - make_fakebin "$d" >/dev/null - head=$(git -C "$d/wt" rev-parse HEAD) - fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" \ - "mode=no-mistakes" - printf 'implementation_completed_at=1\nimplementation_completed_head=%s\n' "$head" >> "$d/state/$id.meta" - FM_FAKE_AXI_STATUS=$(run_passed "fm/$id") - FM_FAKE_BUSY=0 - arm_idle_record "$d/state" "$id" - out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: parked" "passed run without a plan must remain parked" - printf 'validation_generation=plan-1\nvalidation_path=full-no-mistakes\nvalidation_head=%s\n' "$head" >> "$d/state/$id.meta" - out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: parked" "passed run without plan completion must remain parked" - assert_contains "$out" "source: validation-gate" "missing completion must name the validation gate" - printf 'validation_completed_generation=plan-1\nvalidation_completed_path=full-no-mistakes\nvalidation_completed_head=%s\n' "$head" \ - >> "$d/state/$id.meta" - out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: done" "current plan completion must release final done" - pass "run-step done requires current-generation validation completion" + local d out id mode + for mode in no-mistakes direct-PR; do + id="delivery-gate-$mode" + d=$(new_case "$id") + make_repo_on_branch "$d/wt" "fm/$id" + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" "mode=$mode" + printf 'done: PR https://github.com/o/r/pull/1 checks green\n' > "$d/state/$id.status" + if [ "$mode" = no-mistakes ]; then FM_FAKE_AXI_STATUS=$(run_passed "fm/$id"); else FM_FAKE_AXI_STATUS=; fi + FM_FAKE_RUNS_LIST= + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" "$id" + out=$(run_crew_state "$d" "$id") + assert_contains "$out" "state: parked" "$mode done escaped before PR registration" + assert_contains "$out" "source: delivery-gate" "$mode PR wait did not name the delivery gate" + assert_contains "$out" "PR not registered" "$mode PR wait did not say what is missing" + printf 'pr=https://github.com/o/r/pull/1\n' >> "$d/state/$id.meta" + [ "$mode" != no-mistakes ] || printf 'nm_run_id=01RUN\n' >> "$d/state/$id.meta" + out=$(run_crew_state "$d" "$id") + assert_contains "$out" "state: done" "$mode registered PR did not release done" + printf 'untracked\n' > "$d/wt/untracked.txt" + out=$(run_crew_state "$d" "$id") + assert_contains "$out" "state: parked" "$mode dirty worktree escaped final done acceptance" + assert_contains "$out" "worktree is dirty or could not be inspected" "$mode dirty gate omitted its reason" + rm -f "$d/wt/untracked.txt" + done + pass "PR-mode done requires the PR registered by fm-pr-check and a clean worktree" } -test_status_log_done_requires_existing_plan_completion() { +test_local_only_done_requires_clean_task_branch() { reset_fakes - local d out id=status-validation-stage head - d=$(new_case status-validation-stage) - make_repo_on_branch "$d/wt" "fm/$id" + local d out id=delivery-gate-local + d=$(new_case "$id") + make_repo_on_branch "$d/wt" other-branch make_fakebin "$d" >/dev/null - head=$(git -C "$d/wt" rev-parse HEAD) - fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" \ - "mode=direct-PR" - printf 'implementation_completed_at=1\nimplementation_completed_head=%s\n' "$head" >> "$d/state/$id.meta" - printf 'done: PR https://example.test/pull/1\n' > "$d/state/$id.status" + fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" "mode=local-only" + printf 'done: ready in branch fm/%s\n' "$id" > "$d/state/$id.status" FM_FAKE_AXI_STATUS= FM_FAKE_RUNS_LIST= FM_FAKE_BUSY=0 arm_idle_record "$d/state" "$id" out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: parked" "direct-PR done without a plan must remain parked" - printf 'validation_generation=plan-2\nvalidation_path=direct-PR\nvalidation_head=%s\n' "$head" >> "$d/state/$id.meta" + assert_contains "$out" "state: parked" "local-only done on another branch was accepted" + assert_contains "$out" "source: delivery-gate" "local-only branch gate did not name itself" + assert_contains "$out" "branch fm/$id is not checked out" "local-only branch gate did not say what is missing" + git -C "$d/wt" checkout -q -b "fm/$id" out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: parked" "status-log done with an incomplete plan must remain parked" - assert_contains "$out" "source: validation-gate" "status-log completion must name the validation gate" - pass "status-log done requires existing plan completion" + assert_contains "$out" "state: done" "local-only done on the clean task branch was refused" + pass "local-only done requires the clean fm/ branch and no PR" } -test_low_validation_waits_for_pr_completion() { +test_done_time_ask_user_audit_uses_recorded_run() { reset_fakes - local d out id=low-pr-stage head statusbin real_git - d=$(new_case low-pr-stage) + local d out id=done-ask-audit + d=$(new_case "$id") make_repo_on_branch "$d/wt" "fm/$id" make_fakebin "$d" >/dev/null - head=$(git -C "$d/wt" rev-parse HEAD) - fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" \ - "mode=no-mistakes" "implementation_completed_at=1" "implementation_completed_head=$head" \ - "validation_generation=low-plan" "validation_path=receipts-mechanical" "validation_head=$head" - printf 'done: implementation complete\n' > "$d/state/$id.status" + nm_db " + INSERT INTO runs VALUES ('RUN-done-ask'); + INSERT INTO step_results (id, run_id, step_name, step_order, status, findings_json) + VALUES ('sr-done', 'RUN-done-ask', 'review', 3, 'completed', + '{\"findings\":[{\"id\":\"R9\",\"action\":\"ask-user\"}]}'); + INSERT INTO step_rounds (id, step_result_id, round, selection_source, selected_finding_ids, findings_json) + VALUES ('sr-done-1', 'sr-done', 1, 'user_declined', '[]', '{\"findings\":[{\"id\":\"R9\",\"action\":\"ask-user\"}]}'); + " + fm_write_meta "$d/state/$id.meta" "window=fm:fm-$id" "worktree=$d/wt" "kind=ship" "harness=claude" "mode=no-mistakes" \ + "pr=https://github.com/o/r/pull/1" "nm_run_id=RUN-done-ask" + printf 'done: PR https://github.com/o/r/pull/1 checks green\n' > "$d/state/$id.status" FM_FAKE_AXI_STATUS= FM_FAKE_RUNS_LIST= FM_FAKE_BUSY=0 arm_idle_record "$d/state" "$id" out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: parked" "LOW implementation done escaped before PR completion" - assert_contains "$out" "source: validation-gate" "LOW PR wait did not name the validation gate" - printf 'validation_completed_generation=low-plan\nvalidation_completed_path=receipts-mechanical\nvalidation_completed_head=%s\n' "$head" \ - >> "$d/state/$id.meta" - printf 'untracked\n' > "$d/wt/untracked.txt" - out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: parked" "dirty worktree escaped final done acceptance" - assert_contains "$out" "worktree is dirty or could not be inspected" "dirty final gate omitted its reason" - rm -f "$d/wt/untracked.txt" - statusbin="$d/statusbin" - real_git=$(command -v git) - mkdir -p "$statusbin" - cat > "$statusbin/git" < "$d/state/$id.status" out=$(run_crew_state "$d" "$id") - assert_contains "$out" "state: done" "clean LOW PR completion did not release final done" - pass "final done requires a clean inspectable worktree" - pass "LOW validation remains parked until PR completion" + assert_contains "$out" "state: done" "a firstmate-decided finding was still refused: $out" + out=$(NM_HOME="$TMP_ROOT/nm-missing" run_crew_state "$d" "$id") + assert_contains "$out" "state: parked" "unreadable decision data was accepted as done" + assert_contains "$out" "decision evidence could not be read" "unreadable decision data did not say so" + pass "done acceptance applies the ask-user decision audit to the recorded run" } test_fast_modes_skip_no_mistakes_lookup() { @@ -1784,9 +1803,9 @@ test_pipeline_owned_terminal_run_not_exempt test_missing_run_head_falls_back_to_current_state test_ship_done_is_held_until_evidence_is_complete test_ship_done_with_malformed_brief_fails_closed -test_run_step_done_requires_current_plan_completion -test_status_log_done_requires_existing_plan_completion -test_low_validation_waits_for_pr_completion +test_pr_modes_done_requires_registered_pr +test_local_only_done_requires_clean_task_branch +test_done_time_ask_user_audit_uses_recorded_run test_fast_modes_skip_no_mistakes_lookup test_missing_or_malformed_ship_mode_fails_before_run_lookup diff --git a/tests/fm-main-ci-watch.test.sh b/tests/fm-main-ci-watch.test.sh index 561c4cfacf3..6a5a0585d2c 100644 --- a/tests/fm-main-ci-watch.test.sh +++ b/tests/fm-main-ci-watch.test.sh @@ -36,6 +36,20 @@ BASE_PATH=$PATH # gh-axi and gh mocks. The gh mock answers the merge-read calls # fm-pr-merge.sh makes and the CI-watch calls the armed check makes, all from # case-local response files so each test controls the forge's answers. +# A direct-PR task with complete acceptance evidence: fm-pr-check.sh gates every +# ship registration on bin/fm-receipt-check.sh before it records pr=, and these +# cases exercise merge mechanics rather than the handoff gates owned by +# tests/fm-pr-check-handoff.test.sh. +write_evidence_contract() { # [id] + local case_dir=$1 id=${2:-task-x1} + mkdir -p "$case_dir/data/$id" + printf '# Task\nFixture.\n\n# Acceptance criteria\n- AC1: Fixture works.\n\n# Definition of done\nDelivery contract: mode=direct-PR\n' \ + > "$case_dir/data/$id/brief.md" + printf '%s\n' '{"criterion":"AC1","type":"test","outcome":"success","summary":"fixture","result":"passed"}' \ + > "$case_dir/data/$id/evidence.jsonl" + : > "$case_dir/data/$id/.evidence.lock" +} + make_case() { local name=$1 case_dir fakebin case_dir="$TMP_ROOT/$name" @@ -46,7 +60,8 @@ make_case() { "worktree=$case_dir/wt" \ "project=$case_dir/project" \ "kind=ship" \ - "mode=no-mistakes" + "mode=direct-PR" + write_evidence_contract "$case_dir" printf '%s\n' "$case_dir" } @@ -148,6 +163,7 @@ run_merge() { FM_ROOT_OVERRIDE="$ROOT" \ FM_HOME="${FM_TEST_HOME:-$ROOT}" \ FM_STATE_OVERRIDE="$case_dir/state" \ + FM_DATA_OVERRIDE="$case_dir/data" \ FM_TEST_GH_AXI_LOG="$case_dir/gh-axi.log" \ FM_TEST_GH_LOG="$case_dir/gh.log" \ FM_TEST_GH_OUTCOME="$case_dir/github-outcome" \ diff --git a/tests/fm-pr-check-handoff.test.sh b/tests/fm-pr-check-handoff.test.sh new file mode 100755 index 00000000000..0997536e41b --- /dev/null +++ b/tests/fm-pr-check-handoff.test.sh @@ -0,0 +1,269 @@ +#!/usr/bin/env bash +# Behavior tests for the PR-ready handoff gates in bin/fm-pr-check.sh: complete +# acceptance evidence for every ship mode, No-Mistakes run identity proven from +# the pipeline's own status, and the ask-user decision audit at PR-ready. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +PR_CHECK="$ROOT/bin/fm-pr-check.sh" +RECEIPT="$ROOT/bin/fm-receipt.sh" +TMP_ROOT=$(fm_test_tmproot fm-pr-check-handoff) +HOME_DIR="$TMP_ROOT/home" +mkdir -p "$HOME_DIR/data" "$HOME_DIR/state" "$HOME_DIR/config" +fm_git_identity fmtest fmtest@example.invalid +command -v jq >/dev/null 2>&1 || { echo "skip: jq not found"; exit 0; } +command -v python3 >/dev/null 2>&1 || { echo "skip: python3 not found"; exit 0; } + +FAKEBIN=$(fm_fakebin "$TMP_ROOT") +NM_CALL_LOG="$TMP_ROOT/no-mistakes-calls.log" +: > "$NM_CALL_LOG" +cat > "$FAKEBIN/gh" <<'SH' +#!/usr/bin/env bash +case " $* " in + *" headRefOid "*) [ -z "${FM_FAKE_GH_HEAD:-}" ] || printf '%s\n' "$FM_FAKE_GH_HEAD" ;; +esac +SH +cat > "$FAKEBIN/no-mistakes" <> "$NM_CALL_LOG" +[ "\${FM_FAKE_NM_DOWN:-0}" = 0 ] || exit 1 +case "\$*" in + *"axi logs --step ci"*) printf '%s\n' "\${FM_FAKE_NM_CI_LOG:-}" ;; + *) printf '%s\n' "\${FM_FAKE_NM_STATUS:-}" ;; +esac +EOF +chmod +x "$FAKEBIN/gh" "$FAKEBIN/no-mistakes" +export FM_NO_MISTAKES_BIN="$FAKEBIN/no-mistakes" +export FM_GUARD_READ_ONLY=1 + +# The no-mistakes daemon state database is read-only evidence for the +# ask-user decision audit; the suite gets an isolated fixture copy. +NM_DIR="$TMP_ROOT/nm-home" +mkdir -p "$NM_DIR" +python3 - "$NM_DIR" <<'PY' +import os, sqlite3, sys +db = sqlite3.connect(os.path.join(sys.argv[1], "state.sqlite")) +db.executescript(""" +CREATE TABLE runs (id TEXT PRIMARY KEY); +CREATE TABLE step_results ( + id TEXT PRIMARY KEY, run_id TEXT, step_name TEXT, step_order INTEGER, status TEXT, + findings_json TEXT, approval_reason TEXT, override_reason TEXT, skip_reason TEXT +); +CREATE TABLE step_rounds ( + id TEXT PRIMARY KEY, step_result_id TEXT, round INTEGER, selection_source TEXT, + selected_finding_ids TEXT, findings_json TEXT, user_findings_json TEXT +); +""") +db.commit() +PY +export NM_HOME="$NM_DIR" + +nm_db() { # + python3 - "$NM_DIR" "$1" <<'PY' +import os, sqlite3, sys +db = sqlite3.connect(os.path.join(sys.argv[1], "state.sqlite")) +db.executescript(sys.argv[2]) +db.commit() +PY +} + +HEAD_A=0123456789abcdef0123456789abcdef01234567 +HEAD_B=89abcdef0123456789abcdef0123456789abcdef + +nm_status() { # [outcome] [pr] + printf 'run:\n id: "%s"\n branch: %s\n status: %s\n head: %s\n head_sha: %s\n' "$1" "$2" "$4" "${3:0:8}" "$3" + [ -z "${6:-}" ] || printf ' pr: "%s"\n' "$6" + [ -z "${5:-}" ] || printf 'outcome: %s\n' "$5" +} + +make_task() { # -> worktree on fm/ + local id=$1 mode=$2 repo wt + repo="$TMP_ROOT/repo-$id" + wt="$TMP_ROOT/wt-$id" + fm_git_worktree "$repo" "$wt" "fm/$id" + mkdir -p "$HOME_DIR/data/$id" + cat > "$HOME_DIR/data/$id/brief.md" < "$HOME_DIR/data/$id/evidence.jsonl" + : > "$HOME_DIR/data/$id/.evidence.lock" + fm_write_meta "$HOME_DIR/state/$id.meta" "window=fm-$id" "endpoint_task_id=$id" \ + "worktree=$wt" "project=$repo" "kind=ship" "mode=$mode" + : > "$HOME_DIR/state/$id.status" + printf '%s\n' "$wt" +} + +receipt() { # [outcome] + FM_HOME="$HOME_DIR" "$RECEIPT" "$1" "$2" test "evidence for $2" passed --outcome "${3:-success}" >/dev/null \ + || fail "fixture receipt failed for $1/$2" +} + +run_pr_check() { # + PR_OUT=$(FM_HOME="$HOME_DIR" PATH="$FAKEBIN:$PATH" "$PR_CHECK" "$1" "$2" 2>&1) + PR_RC=$? +} + +assert_not_armed() { #