From 342f69efab2a15327c2f64be981f089ee6c4690b Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 13:53:18 +0200 Subject: [PATCH 01/11] Strengthen generated delivery contracts --- bin/fm-dod-lib.sh | 12 ++++++-- tests/fm-brief.test.sh | 56 ++++++++++++++++++++++++++++++++++ tests/fm-task-delivery.test.sh | 8 ++++- 3 files changed, 73 insertions(+), 3 deletions(-) diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index 07a7b46e242..08583f7e70e 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -198,9 +198,14 @@ fm_dod_block() { # # Definition of done Delivery contract: mode=direct-PR This task ships **direct-PR**: you raise the PR yourself, without the no-mistakes pipeline. +Do not run /no-mistakes unless firstmate explicitly instructs you to change this task's delivery path. The task is complete only when committed on your branch. -When it is implemented and committed, push your branch and open a PR with \`gh-axi\`, then append \`done: PR {url}\` to the status file and stop. -Do NOT run /no-mistakes. The configured merge authority decides whether to merge the PR; firstmate relays the outcome. +When it is implemented and committed, push your branch and open a PR with \`gh-axi\`. +If a push, PR creation, or PR verification fails, diagnose the forge failure first, including the reported authentication, remote, branch, or API error; do not use no-mistakes as a workaround. +Before the final status, verify with \`gh-axi\` that the branch was actually pushed and that the forge reports a full \`https://...\` PR URL for that branch. +A local commit, an attempted push, a bare PR number, or an inferred URL is not done. +Only after those checks append \`done: PR {url}\` to the status file and stop. +The configured merge authority decides whether to merge the PR; firstmate relays the outcome. EOF ;; local-only) @@ -237,6 +242,9 @@ So background the drive call and poll \`no-mistakes axi status\` from a separate Where a harness's own command limit is not established, assume it bounds commands and use that same background-and-poll shape. A killed or timed-out call is never evidence the daemon died: the daemon accepts your response immediately and runs the round in the background, so the call was only ever waiting for a read while the run kept working. Reattach and keep going rather than reporting the pipeline blocked; rule 7 owns the checks that decide when a pipeline block is real. +After every \`no-mistakes axi respond\`, continue in the same turn with bounded calls to the structured \`no-mistakes axi status\` interface until the attributed run changes step, reaches a terminal outcome, or presents a genuine ask-user decision. +An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run. +Never end your turn or promise to resume or check later while structured status shows that validation is active. Two firstmate-specific rules layer on top of that guidance: - ask-user findings are never yours to answer: escalate to firstmate using rule 6's ask-user format and stop. diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 5467b5cbdca..ea49722aff6 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -372,6 +372,60 @@ test_no_mistakes_dod_wording() { pass "fm-brief.sh: no-mistakes DOD keeps its apostrophe prose and bans --yes outright" } +test_active_no_mistakes_validation_cannot_be_deferred() { + local home id brief + home="$TMP_ROOT/active-validation-home" + mkdir -p "$home/data" + id="brief-active-validation-c1" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --mode no-mistakes >/dev/null 2>&1 + brief="$home/data/$id/brief.md" + + assert_grep "After every \`no-mistakes axi respond\`, continue in the same turn" "$brief" \ + "no-mistakes brief did not require same-turn continuation after a gate response" + assert_grep "bounded calls to the structured \`no-mistakes axi status\` interface" "$brief" \ + "no-mistakes brief did not require bounded structured status polling" + assert_grep "until the attributed run changes step, reaches a terminal outcome, or presents a genuine ask-user decision" "$brief" \ + "no-mistakes brief did not define the only status-polling stop conditions" + assert_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active." "$brief" \ + "no-mistakes brief still permits deferring an active validation run" + pass "fm-brief.sh: active no-mistakes validation continues in the same turn through the next real transition" +} + +test_direct_pr_requires_forge_proof_and_diagnosis() { + local home id brief scout charter local_brief + home="$TMP_ROOT/direct-pr-proof-home" + mkdir -p "$home/data" + id="brief-direct-pr-proof-c2" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --mode direct-PR >/dev/null 2>&1 + brief="$home/data/$id/brief.md" + + assert_grep "Do not run /no-mistakes unless firstmate explicitly instructs you to change this task's delivery path." "$brief" \ + "direct-PR brief did not forbid an unrequested no-mistakes run" + assert_grep "diagnose the forge failure first" "$brief" \ + "direct-PR brief did not require forge-first diagnosis" + assert_grep "verify with \`gh-axi\` that the branch was actually pushed" "$brief" \ + "direct-PR brief did not require proof of the remote branch" + assert_grep "full \`https://...\` PR URL" "$brief" \ + "direct-PR brief did not require a verified full PR URL" + assert_grep "A local commit, an attempted push, a bare PR number, or an inferred URL is not done." "$brief" \ + "direct-PR brief permits an unproved final announcement" + + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" unaffected-scout some-proj --scout >/dev/null 2>&1 + scout="$home/data/unaffected-scout/brief.md" + FM_HOME="$home" FM_SECONDMATE_CHARTER='Supervise assigned work.' \ + "$ROOT/bin/fm-brief.sh" unaffected-charter --secondmate --no-projects >/dev/null 2>&1 + charter="$home/data/unaffected-charter/brief.md" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" unaffected-local some-proj --mode local-only >/dev/null 2>&1 + local_brief="$home/data/unaffected-local/brief.md" + for unaffected in "$scout" "$charter" "$local_brief"; do + assert_no_grep "diagnose the forge failure first" "$unaffected" \ + "an unaffected scaffold received the direct-PR forge contract" + assert_no_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active." "$unaffected" \ + "an unaffected scaffold received the no-mistakes active-run contract" + done + pass "fm-brief.sh: direct-PR completion requires forge diagnosis, a pushed branch, and a verified full URL" +} + test_ask_user_escalation_format() { local home id brief mode other_id other_brief home="$TMP_ROOT/ask-user-home" @@ -878,6 +932,8 @@ test_ship_mode_is_explicit_not_registry test_delivery_flags_are_refused_where_they_do_not_apply test_faster_paths_use_configured_authority_without_stacked_review test_no_mistakes_dod_wording +test_active_no_mistakes_validation_cannot_be_deferred +test_direct_pr_requires_forge_proof_and_diagnosis test_ask_user_escalation_format test_ship_project_memory_wording test_herdr_lab_contract_is_explicit_and_complete diff --git a/tests/fm-task-delivery.test.sh b/tests/fm-task-delivery.test.sh index bdace4b501e..9139ce9ed84 100755 --- a/tests/fm-task-delivery.test.sh +++ b/tests/fm-task-delivery.test.sh @@ -380,6 +380,8 @@ STUB "promoted no-mistakes worker did not receive the --yes prohibition" assert_grep "It is banned fleet-wide" "$payload" \ "promoted no-mistakes worker did not receive the fleet-wide ban wording" + assert_grep "After every \`no-mistakes axi respond\`, continue in the same turn" "$payload" \ + "promoted no-mistakes worker can still defer an active validation run" payload="$TMP_ROOT/promote-dod/payload-promote-dod-direct-pr" assert_grep "supersede the scout delivery rules and report-based Definition of done" "$payload" \ @@ -388,8 +390,12 @@ STUB "promoted worker lost the scout protocols and safety rules that still apply" # The faster paths keep their own contracts rather than inheriting the pipeline's. - assert_grep "Do NOT run /no-mistakes" "$payload" \ + assert_grep "Do not run /no-mistakes unless firstmate explicitly instructs you to change this task's delivery path." "$payload" \ "promoted direct-PR worker lost its no-pipeline contract" + assert_grep "verify with \`gh-axi\` that the branch was actually pushed" "$payload" \ + "promoted direct-PR worker lost its pushed-branch proof" + assert_grep "diagnose the forge failure first" "$payload" \ + "promoted direct-PR worker lost its forge-first diagnosis contract" assert_grep "Do NOT push, do NOT open a PR, do NOT merge" "$TMP_ROOT/promote-dod/payload-promote-dod-local-only" \ "promoted local-only worker lost its no-remote contract" assert_no_grep "no-mistakes axi respond" "$TMP_ROOT/promote-dod/payload-promote-dod-direct-pr" \ From f910458c4f6e6173ce0757078f472f0395b54313 Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 14:03:36 +0200 Subject: [PATCH 02/11] no-mistakes(review): Add rule-7 block escape, drop duplicate direct-PR line --- bin/fm-dod-lib.sh | 5 ++--- tests/fm-brief.test.sh | 8 +++----- 2 files changed, 5 insertions(+), 8 deletions(-) diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index 08583f7e70e..b3c304b3dda 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -203,7 +203,6 @@ The task is complete only when committed on your branch. When it is implemented and committed, push your branch and open a PR with \`gh-axi\`. If a push, PR creation, or PR verification fails, diagnose the forge failure first, including the reported authentication, remote, branch, or API error; do not use no-mistakes as a workaround. Before the final status, verify with \`gh-axi\` that the branch was actually pushed and that the forge reports a full \`https://...\` PR URL for that branch. -A local commit, an attempted push, a bare PR number, or an inferred URL is not done. Only after those checks append \`done: PR {url}\` to the status file and stop. The configured merge authority decides whether to merge the PR; firstmate relays the outcome. EOF @@ -242,9 +241,9 @@ So background the drive call and poll \`no-mistakes axi status\` from a separate Where a harness's own command limit is not established, assume it bounds commands and use that same background-and-poll shape. A killed or timed-out call is never evidence the daemon died: the daemon accepts your response immediately and runs the round in the background, so the call was only ever waiting for a read while the run kept working. Reattach and keep going rather than reporting the pipeline blocked; rule 7 owns the checks that decide when a pipeline block is real. -After every \`no-mistakes axi respond\`, continue in the same turn with bounded calls to the structured \`no-mistakes axi status\` interface until the attributed run changes step, reaches a terminal outcome, or presents a genuine ask-user decision. +After every \`no-mistakes axi respond\`, continue in the same turn with bounded calls to the structured \`no-mistakes axi status\` interface until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block. An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run. -Never end your turn or promise to resume or check later while structured status shows that validation is active. +Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block. Two firstmate-specific rules layer on top of that guidance: - ask-user findings are never yours to answer: escalate to firstmate using rule 6's ask-user format and stop. diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index ea49722aff6..17f4fa410f9 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -384,9 +384,9 @@ test_active_no_mistakes_validation_cannot_be_deferred() { "no-mistakes brief did not require same-turn continuation after a gate response" assert_grep "bounded calls to the structured \`no-mistakes axi status\` interface" "$brief" \ "no-mistakes brief did not require bounded structured status polling" - assert_grep "until the attributed run changes step, reaches a terminal outcome, or presents a genuine ask-user decision" "$brief" \ + assert_grep "until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block" "$brief" \ "no-mistakes brief did not define the only status-polling stop conditions" - assert_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active." "$brief" \ + assert_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block." "$brief" \ "no-mistakes brief still permits deferring an active validation run" pass "fm-brief.sh: active no-mistakes validation continues in the same turn through the next real transition" } @@ -407,8 +407,6 @@ test_direct_pr_requires_forge_proof_and_diagnosis() { "direct-PR brief did not require proof of the remote branch" assert_grep "full \`https://...\` PR URL" "$brief" \ "direct-PR brief did not require a verified full PR URL" - assert_grep "A local commit, an attempted push, a bare PR number, or an inferred URL is not done." "$brief" \ - "direct-PR brief permits an unproved final announcement" FM_HOME="$home" "$ROOT/bin/fm-brief.sh" unaffected-scout some-proj --scout >/dev/null 2>&1 scout="$home/data/unaffected-scout/brief.md" @@ -420,7 +418,7 @@ test_direct_pr_requires_forge_proof_and_diagnosis() { for unaffected in "$scout" "$charter" "$local_brief"; do assert_no_grep "diagnose the forge failure first" "$unaffected" \ "an unaffected scaffold received the direct-PR forge contract" - assert_no_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active." "$unaffected" \ + assert_no_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block." "$unaffected" \ "an unaffected scaffold received the no-mistakes active-run contract" done pass "fm-brief.sh: direct-PR completion requires forge diagnosis, a pushed branch, and a verified full URL" From 7e34d2e8b0e102bb1dfb7aed58983d1532b4cd4a Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 14:30:37 +0200 Subject: [PATCH 03/11] no-mistakes(review): Absorb live no-mistakes runs in busy-turn bound --- bin/fm-classify-lib.sh | 36 ++++++++++++--- bin/fm-dod-lib.sh | 1 - bin/fm-watch.sh | 20 +++++++- tests/fm-watch-triage.test.sh | 86 +++++++++++++++++++++++++++++++++++ 4 files changed, 134 insertions(+), 9 deletions(-) diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 8bd4fe746ad..aac919bceac 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -1744,6 +1744,22 @@ status_span_has_actionable() { # status_span_first_actionable_record "$1" "${2:-0}" > /dev/null } +# Read bin/fm-crew-state.sh's one authoritative current-state line +# ("state: · source: · ") and print it as " ", +# the two fields every absorb decision is made from. Fails (prints nothing) when +# no id is given or the read is missing/unparseable, so a caller cannot mistake an +# unreadable crew for a classified one. Not a pure read: see crew_absorb_class. +# FM_CREW_STATE_BIN lets tests stub the verdict. +crew_state_verdict() { # + local id=$1 line state src= + [ -n "$id" ] || return 1 + line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || true + case "$line" in state:*) ;; *) return 1 ;; esac + state=${line#state: }; state=${state%% *} + case "$line" in *"source: "*) src=${line#*source: }; src=${src%% *} ;; esac + printf '%s %s' "$state" "$src" +} + # Classify WHY an idle/stale crew MIGHT be safely absorbed instead of surfaced, # from bin/fm-crew-state.sh's one authoritative current-state line # ("state: · source: · "). Prints exactly one token: @@ -1761,14 +1777,11 @@ status_span_has_actionable() { # # run it only on no-verb signal and first-sighting stale paths, never every wake. # FM_CREW_STATE_BIN lets tests stub the verdict. crew_absorb_class() { # - local id=$1 line state src - [ -n "$id" ] || { printf 'none'; return; } - line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || true - case "$line" in state:*) ;; *) printf 'none'; return ;; esac - state=${line#state: }; state=${state%% *} + local verdict state src + verdict=$(crew_state_verdict "$1") || { printf 'none'; return; } + state=${verdict%% *}; src=${verdict##* } if [ "$state" = paused ]; then printf 'paused'; return; fi if [ "$state" = working ]; then - src=${line#*source: }; src=${src%% *} case "$src" in run-step|pane) printf 'working'; return ;; esac fi printf 'none' @@ -1795,6 +1808,17 @@ crew_is_paused() { # [ "$(crew_absorb_class "$1")" = paused ] } +# 0 iff crew is working BECAUSE an attributed no-mistakes run-step is live - +# the run-step half of crew_absorb_class's `working`, without the busy-pane half. +# Callers that already hold a busy verdict need this narrower proof: a busy pane +# cannot also be its own bound. Attribution is fm-crew-state.sh's (branch AND code +# identity, or pipeline-owned custody), so a run record that no longer matches this +# worktree, a terminal run, and a daemon an explicit probe proves down all report +# something other than working/run-step and are NOT a live run. +crew_run_step_is_live() { # + [ "$(crew_state_verdict "$1")" = "working run-step" ] +} + # Directories excluded from the worktree write probe below, and the depth it walks. # The excluded set is everything a supervisor read or a package manager can write # without the crew doing any work - .git first, so firstmate's own read-only git diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index b3c304b3dda..6dff89705d3 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -242,7 +242,6 @@ Where a harness's own command limit is not established, assume it bounds command A killed or timed-out call is never evidence the daemon died: the daemon accepts your response immediately and runs the round in the background, so the call was only ever waiting for a read while the run kept working. Reattach and keep going rather than reporting the pipeline blocked; rule 7 owns the checks that decide when a pipeline block is real. After every \`no-mistakes axi respond\`, continue in the same turn with bounded calls to the structured \`no-mistakes axi status\` interface until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block. -An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run. Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block. Two firstmate-specific rules layer on top of that guidance: diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index f4246480ff5..7b98f750e9d 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -214,7 +214,8 @@ STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provabl # non-busy stale - so it escalates via the existing stale reason, escalation # counter, and demand-deep-inspection marker for human inspection only, never an # automatic interrupt, signal, or restart - unless the crew declared the wait -# itself, which takes the long pause cadence instead. A completed turn touches +# itself, which takes the long pause cadence instead, or an attributed +# no-mistakes run is still live, which is absorbed as the active wait it is. A completed turn touches # turn-ended and resets the age. Set generously above any legitimate interval # between completed turns, including long tool calls, builds, or test runs. BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} @@ -947,7 +948,15 @@ handle_paused_stale() { # # A busy pane past BUSY_TURN_MAX_SECS is normally a wedge suspect because a hung # foreground call can hide behind a busy signature. A `paused:` declaration or # verified captain-held transfer instead identifies that live foreground call as -# the expected external wait. The caller has already confirmed liveness through +# the expected external wait, and so does an attributed no-mistakes run that is +# still live (crew_run_step_is_live): a validating worker holds ONE turn open for +# the whole run by contract, so its completed-turn age is expected to cross the +# bound, and the run-step is the authoritative evidence that the wait is real +# work rather than a wedge. That absorber is deliberately the narrow run-step +# proof, never crew_absorb_class's `working`, whose busy-pane half would let the +# very pane under test vouch for itself. A run record no longer attributed to +# this worktree, a terminal run, and a proven-down daemon are all NOT live, so +# stale run state still escalates on the normal wedge cadence. The caller has already confirmed liveness through # the busy verdict, so this exception does not suppress undeclared wedges or # alter the separate non-busy classification. handle_paused_stale keeps the # exception bounded by re-surfacing it once per PAUSE_RESURFACE_SECS. Away mode @@ -992,6 +1001,13 @@ busy_turn_bound_check() { # "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/validating.meta" + record_pi_busy "$state" validating + printf 'working: validating\n' > "$state/validating.status" + sig=$(seen_sig "$state/validating.status"); printf '%s' "$sig" > "$state/.seen-validating_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + touch -t 200001010000 "$state/validating.turn-ended" + prime_turnend_seen "$state/validating.turn-ended" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "a busy pane with a live attributed run-step was wedge-escalated: $(cat "$out")" + fi + [ ! -s "$out" ] || fail "a busy pane with a live attributed run-step printed a wake reason" + [ ! -e "$state/.stale-since-$key" ] || fail "a live attributed run-step still started a wedge timer" + [ ! -e "$state/.wedge-escalations-$key" ] || fail "a live attributed run-step still counted a wedge escalation" + reap "$pid" + pass "a busy worker whose attributed no-mistakes run is still live is absorbed past the completed-turn bound" +} + +# The complement of the absorb above, on the same fixture shape: run state that +# is no longer live - a terminal or unattributed record, or a daemon an explicit +# probe proves down (both reported by fm-crew-state.sh as something other than +# working/run-step) - is exactly the case the bound exists for, so it must still +# reach the wedge escalation with its stale reason and escalation counter. +test_busy_pane_with_dead_run_state_still_escalates_past_turn_age_bound() { + local dir state fakebin out capture_file window key sig pid + dir=$(make_case busy-dead-run-state); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-dead-daemon" + printf 'Working... (4210.6s)' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/dead-daemon.meta" + record_pi_busy "$state" dead-daemon + printf 'working: validating\n' > "$state/dead-daemon.status" + sig=$(seen_sig "$state/dead-daemon.status"); printf '%s' "$sig" > "$state/.seen-dead-daemon_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + touch -t 200001010000 "$state/dead-daemon.turn-ended" + prime_turnend_seen "$state/dead-daemon.turn-ended" + + # Phase A: the stale run record cannot vouch for the pane, so the bound starts + # the ordinary wedge timer exactly as it does with no run at all. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_FAKE_CREW_STATE='state: unknown · source: none · no-mistakes daemon unreachable; last ledger record failed - unverified' \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "a busy pane with dead run state escalated before the wedge threshold: $(cat "$out")" + fi + [ -s "$state/.stale-since-$key" ] || fail "a busy pane with dead run state did not start a wedge timer" + reap "$pid" + ack_stopped_cycle "$state" || fail "could not acknowledge the intentional dead-run-state phase-A stop" + + # Phase B: past the escalation threshold it wedge-escalates for human inspection. + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_FAKE_CREW_STATE='state: unknown · source: none · no-mistakes daemon unreachable; last ledger record failed - unverified' \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || fail "a busy pane with dead run state did not wedge-escalate past the turn-age bound" + grep -F "stale: $window" "$out" >/dev/null || fail "dead run state escalation did not print the stale wake" + grep -F "possible wedge" "$out" >/dev/null || fail "dead run state escalation did not flag a possible wedge" + pass "a busy worker whose no-mistakes run state is stale or daemon-down still escalates past the completed-turn bound" +} + # --- declared pause + busy pane: the busy-turn bound must honor the declaration # A single foreground call can keep a declared external wait semantically busy # past the completed-turn bound, bypassing the ordinary stale-pause path. @@ -4432,6 +4516,8 @@ test_busy_pane_stable_hash_escalates_past_turn_age_bound test_busy_pane_changing_hash_escalates_past_turn_age_bound test_busy_pane_turn_end_touch_resets_age test_busy_pane_repeated_escalation_reaches_demand_deep_inspection +test_busy_pane_with_live_run_step_is_absorbed_past_turn_age_bound +test_busy_pane_with_dead_run_state_still_escalates_past_turn_age_bound test_busy_pane_default_turn_age_bound_is_3600s test_busy_declared_pause_is_rechecked_not_wedge_escalated test_afk_busy_declared_pause_hands_off_plain_stale From a069b03c3dedcddd5247b27b7ade5627ab091f49 Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 15:19:31 +0200 Subject: [PATCH 04/11] no-mistakes(review): Require proven-live daemon before deferring busy-turn wedge --- bin/fm-classify-lib.sh | 34 +++++--- bin/fm-crew-state.sh | 12 +-- bin/fm-nm-run-lib.sh | 9 +++ bin/fm-watch.sh | 103 ++++++++++++++--------- tests/fm-watch-triage.test.sh | 148 ++++++++++++++++++++++++---------- tests/wake-helpers.sh | 24 ++++++ 6 files changed, 230 insertions(+), 100 deletions(-) diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index aac919bceac..a3508696307 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -61,6 +61,9 @@ case $- in *u*) _fm_classify_nounset=on ;; *) _fm_classify_nounset=off ;; esac # shellcheck source=bin/fm-timeout-lib.sh # shellcheck disable=SC1091 . "$_FM_CLASSIFY_LIB_DIR/fm-timeout-lib.sh" +# shellcheck source=bin/fm-nm-run-lib.sh +# shellcheck disable=SC1091 +. "$_FM_CLASSIFY_LIB_DIR/fm-nm-run-lib.sh" [ "$_fm_classify_nounset" = on ] || set +u unset _fm_classify_nounset @@ -1808,15 +1811,28 @@ crew_is_paused() { # [ "$(crew_absorb_class "$1")" = paused ] } -# 0 iff crew is working BECAUSE an attributed no-mistakes run-step is live - -# the run-step half of crew_absorb_class's `working`, without the busy-pane half. -# Callers that already hold a busy verdict need this narrower proof: a busy pane -# cannot also be its own bound. Attribution is fm-crew-state.sh's (branch AND code -# identity, or pipeline-owned custody), so a run record that no longer matches this -# worktree, a terminal run, and a daemon an explicit probe proves down all report -# something other than working/run-step and are NOT a live run. -crew_run_step_is_live() { # - [ "$(crew_state_verdict "$1")" = "working run-step" ] +# 0 only on POSITIVE proof that crew 's no-mistakes validation is running +# right now, from two facts that must BOTH hold: +# - fm-crew-state.sh attributes a working run-step to this crew (branch AND code +# identity, or pipeline-owned custody), which rules out a terminal run and a +# record that no longer belongs to this worktree; +# - fm_nm_daemon_is_alive proves the shared daemon is up. +# The second is not redundant: a run record left at running/fixing is never +# advanced after the daemon exits, so the record alone reports a live run for a +# validation nothing is executing - the same stale-record hazard rule 7 of the +# generated brief makes crews check before appending `blocked:`. Every failure is +# a negative answer, so a missing worktree, an unreadable verdict, and a probe +# that times out all read as NOT live. +# The busy-pane half of crew_absorb_class's `working` is deliberately excluded: a +# caller that already holds a busy verdict cannot let that pane vouch for itself. +# Two bounded subprocesses per call, so callers must run it at most once per +# STALE_ESCALATE_SECS - never per poll (see crew_absorb_class). +crew_nm_run_is_provably_live() { # + local id=$1 state=$2 wt + [ "$(crew_state_verdict "$id")" = "working run-step" ] || return 1 + wt=$(grep '^worktree=' "$state/$id.meta" 2>/dev/null | tail -1 | cut -d= -f2- || true) + [ -n "$wt" ] && [ -d "$wt" ] || return 1 + fm_nm_daemon_is_alive "$wt" "${FM_CREW_STATE_NM_TIMEOUT:-10}" } # Directories excluded from the worktree write probe below, and the depth it walks. diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index f3a99c3e3e5..fc54687952b 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -453,15 +453,11 @@ nm_reclassify_failed_run_as_held_green() { return 0 } -# 0 when an explicit probe proves the shared daemon down: `no-mistakes daemon -# status` is the canonical down-probe (the same one fm-brief.sh hands crews -# before a blocked append) and exits non-zero when the daemon is not running. -# Bounded like every other CLI call; a probe that fails for any reason - -# refused socket, timeout, non-zero answer - means the daemon is not provably -# up, which is the only fact the coarse fallback needs. +# 0 when an explicit probe proves the shared daemon down - the negation of +# fm_nm_daemon_is_alive in bin/fm-nm-run-lib.sh, which owns that one probe for +# every caller that needs daemon liveness. nm_daemon_probe_down() { - fm_nm_run_checked "$WT" "$NM_TIMEOUT" daemon status >/dev/null || return 0 - return 1 + ! fm_nm_daemon_is_alive "$WT" "$NM_TIMEOUT" } nm_ci_step_status() { diff --git a/bin/fm-nm-run-lib.sh b/bin/fm-nm-run-lib.sh index 9a04004e391..328ba28ef42 100644 --- a/bin/fm-nm-run-lib.sh +++ b/bin/fm-nm-run-lib.sh @@ -38,6 +38,15 @@ fm_nm_run() { # fm_nm_run_checked "$@" || true } +# 0 only when a bounded `no-mistakes daemon status` PROVES the shared daemon up. +# The canonical liveness probe (the same one fm-brief.sh hands crews before a +# blocked append), stated positively: any failure - refused socket, timeout, +# missing binary, non-zero answer - means the daemon is not provably up, so every +# caller that needs daemon liveness fails closed on the same one fact. +fm_nm_daemon_is_alive() { # + fm_nm_run_checked "$1" "$2" daemon status >/dev/null +} + fm_nm_trim() { local s=${1:-} s="${s#"${s%%[![:space:]]*}"}" diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 7b98f750e9d..9b69ca5cd7c 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -214,8 +214,9 @@ STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provabl # non-busy stale - so it escalates via the existing stale reason, escalation # counter, and demand-deep-inspection marker for human inspection only, never an # automatic interrupt, signal, or restart - unless the crew declared the wait -# itself, which takes the long pause cadence instead, or an attributed -# no-mistakes run is still live, which is absorbed as the active wait it is. A completed turn touches +# itself, which takes the long pause cadence instead, or a no-mistakes validation +# is proven live at the escalation threshold, which is deferred onto that same +# bounded cadence for as long as the proof keeps holding. A completed turn touches # turn-ended and resets the age. Set generously above any legitimate interval # between completed turns, including long tool calls, builds, or test runs. BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} @@ -794,8 +795,9 @@ FM_WEDGE_DEMAND_INSPECT_COUNT=${FM_WEDGE_DEMAND_INSPECT_COUNT:-3} # once past PAUSE_RESURFACE_SECS the pane wakes once per window rather than every # poll. An optional binds that cadence to its current declaration; callers # without a scoped declaration keep the timestamp body. Shared by the -# declared-pause absorb and the worktree-write deferral so the two cadences cannot -# drift apart; each caller owns its own marker and reason. +# declared-pause absorb and both wedge deferrals (worktree-write and live +# no-mistakes run) so their cadences cannot drift apart; each caller owns its own +# marker and reason. # Returns without waking while either the absorb or the throttle is inside the # window; wake() itself exits the cycle, exactly as it does inline. resurface_absorbed() { # [scope] @@ -836,12 +838,39 @@ wedge_defer_writing() { # triage_log "absorbed $label (worktree written since the idle window opened, idle ${age}s): $win" } -# Drop a window's write-deferral chain wherever its stale bookkeeping resets, so -# the bounded re-surface cadence is measured from the CURRENT quiet stretch and a -# long-finished one cannot make the next deferral resurface immediately. -clear_write_tracking() { # +# Defer ONE wedge escalation for a pane whose no-mistakes validation is provably +# live (crew_nm_run_is_provably_live: an attributed working run-step AND a daemon +# proven up). This is the busy-turn bound's second real wait: the crew has not +# completed a turn because the generated Definition of done forbids ending the +# turn while its validation is active, so the pipeline - not the pane - is the +# evidence. Deliberately the same DEFERRAL shape as the worktree-write one, never +# a cancellation: the idle timer restarts so the next window re-proves both facts +# from scratch, and a .nmrun-since- chain ages the whole deferral so the pane +# still re-surfaces once every PAUSE_RESURFACE_SECS through resurface_absorbed. A +# run that stops advancing therefore cannot stay invisible, and the moment the +# record goes terminal, loses attribution, or the daemon dies, the next threshold +# falls straight through to the unchanged escalation ladder. The escalation +# counter is left alone for the same reason wedge_defer_writing leaves it. +wedge_defer_live_run() { # + local win=$1 since_file=$2 label=$3 age=$4 key rsf rage + key=$(window_key "$win") + rsf="$STATE/.nmrun-since-$key" + [ -e "$rsf" ] || date +%s > "$rsf" + rage=$(age_of "$rsf") + date +%s > "$since_file" + resurface_absorbed "$win" "$STATE/.nmrun-resurfaced-$key" "$rage" \ + "stale: $win (idle ${age}s, no-mistakes validation proven live for ${rage}s, rechecked on a long cadence not a wedge; confirm the run is still advancing)" + triage_log "absorbed $label (attributed no-mistakes run proven live, idle ${age}s): $win" +} + +# Drop a window's deferral chains - worktree-write and live-run alike - wherever +# its stale bookkeeping resets, so the bounded re-surface cadence is measured from +# the CURRENT quiet stretch and a long-finished chain cannot make the next +# deferral resurface immediately. +clear_defer_tracking() { # local key=$1 - rm -f "$STATE/.writing-since-$key" "$STATE/.writing-resurfaced-$key" + rm -f "$STATE/.writing-since-$key" "$STATE/.writing-resurfaced-$key" \ + "$STATE/.nmrun-since-$key" "$STATE/.nmrun-resurfaced-$key" } # Repeat-poll wedge-timer bookkeeping for an already-classified stale hash @@ -855,14 +884,14 @@ clear_write_tracking() { # # The worktree write probe runs ONLY here, inside the at-threshold branch that is # about to escalate: at most one bounded walk per window per STALE_ESCALATE_SECS, # never per poll. -wedge_timer_check() { # - local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 since age n reason +wedge_timer_check() { # [defer-live-run] + local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 live_run=${6-} since age n reason since=$(cat "$since_file" 2>/dev/null || true) case "$since" in ''|*[!0-9]*) # Publish the repaired timer only after its old write-deferral chain is # gone, so observers cannot mistake a new idle window for the old chain. - clear_write_tracking "$(window_key "$win")" + clear_defer_tracking "$(window_key "$win")" date +%s > "$since_file" triage_log "absorbed $label timer reset: $win" ;; @@ -873,6 +902,10 @@ wedge_timer_check() { # /dev/null || echo 0) + 1 )) echo "$n" > "$escalation_file" reason="stale: $win (idle ${age}s, possible wedge, escalation $n)" @@ -881,7 +914,7 @@ wedge_timer_check() { # printf '%s' "$h" > "$STATE/.stale-$key" : > "$STATE/.paused-$key" rm -f "$STATE/.stale-since-$key" "$STATE/.wedge-escalations-$key" - clear_write_tracking "$key" + clear_defer_tracking "$key" statusf="$STATE/$task.status" mtime=$(stat_mtime "$statusf") case "$mtime" in ''|*[!0-9]*) mtime=$(date +%s) ;; esac @@ -948,15 +981,14 @@ handle_paused_stale() { # # A busy pane past BUSY_TURN_MAX_SECS is normally a wedge suspect because a hung # foreground call can hide behind a busy signature. A `paused:` declaration or # verified captain-held transfer instead identifies that live foreground call as -# the expected external wait, and so does an attributed no-mistakes run that is -# still live (crew_run_step_is_live): a validating worker holds ONE turn open for -# the whole run by contract, so its completed-turn age is expected to cross the -# bound, and the run-step is the authoritative evidence that the wait is real -# work rather than a wedge. That absorber is deliberately the narrow run-step -# proof, never crew_absorb_class's `working`, whose busy-pane half would let the -# very pane under test vouch for itself. A run record no longer attributed to -# this worktree, a terminal run, and a proven-down daemon are all NOT live, so -# stale run state still escalates on the normal wedge cadence. The caller has already confirmed liveness through +# the expected external wait. A live no-mistakes validation is the other real +# wait - a validating worker holds ONE turn open for the whole run by contract, +# so its completed-turn age is expected to cross the bound - but that evidence is +# NOT read here: it is handed to wedge_timer_check as `defer-live-run`, which +# consults it only in the branch that is about to escalate. That keeps the two +# bounded no-mistakes subprocesses to at most one pair per window per +# STALE_ESCALATE_SECS instead of one pair per poll, and keeps the outcome a +# bounded deferral rather than an unbounded silence. The caller has already confirmed liveness through # the busy verdict, so this exception does not suppress undeclared wedges or # alter the separate non-busy classification. handle_paused_stale keeps the # exception bounded by re-surfacing it once per PAUSE_RESURFACE_SECS. Away mode @@ -989,7 +1021,7 @@ busy_turn_bound_check() { # /dev/null || true)" != "$declared" ]; then fm_wake_append stale "$win" "stale: $win" || exit 1 @@ -1001,14 +1033,7 @@ busy_turn_bound_check() { # # recheck, and re-surface throttle - can still reset the per-hash half alone. clear_stale_hash_tracking() { # local key=$1 - clear_write_tracking "$key" + clear_defer_tracking "$key" rm -f "$STATE/.stale-$key" "$STATE/.stale-since-$key" "$STATE/.wedge-escalations-$key" } @@ -1228,7 +1253,7 @@ surface_nonterminal_stale() { # fi printf '%s' "$h" > "$STATE/.stale-$key" rm -f "$STATE/.stale-since-$key" - clear_write_tracking "$key" + clear_defer_tracking "$key" if [ "$declared" -eq 0 ]; then : > "$STATE/.paused-$key" date +%s > "$STATE/.paused-rechecked-$key" @@ -2123,7 +2148,7 @@ EOF if crew_is_provably_working "$(window_to_task "$w" "$STATE")"; then printf '%s' "$h" > "$sf" date +%s > "$ssf" - clear_write_tracking "$key" + clear_defer_tracking "$key" triage_log "absorbed stale (provably working, overriding a stale captain-relevant status): $w" elif captain_call_stale_bound "$key" "$task"; then # The line is captain-relevant and stays so, but the backlog says @@ -2135,14 +2160,14 @@ EOF # here as it already was after a first terminal alarm. printf '%s' "$h" > "$sf" rm -f "$ssf" - clear_write_tracking "$key" + clear_defer_tracking "$key" triage_log "absorbed stale (open captain call already surfaced for this status): $w" else fm_wake_append stale "$w" "stale: $w" || exit 1 stale_wait_record "$key" printf '%s' "$h" > "$sf" rm -f "$ssf" - clear_write_tracking "$key" + clear_defer_tracking "$key" stale_status="$STATE/$(window_to_task "$w" "$STATE").status" stale_record=$(status_span_first_actionable_record "$stale_status" 0) case $? in @@ -2219,7 +2244,7 @@ EOF busy_turn_bound_check "$w" "$task" "$h" "$ssf" "$ewf" && paused_bound=0 else rm -f "$ssf" "$ewf" - clear_write_tracking "$key" + clear_defer_tracking "$key" fi # A busy pane normally means real work resumed, so stale pause bookkeeping # is cleared - but not in the same poll the declared-pause cadence just @@ -2237,7 +2262,7 @@ EOF busy_turn_bound_check "$w" "$task" "$h" "$ssf" "$ewf" && paused_bound=0 else rm -f "$ssf" "$ewf" - clear_write_tracking "$key" + clear_defer_tracking "$key" fi task=$(window_to_task "$w" "$STATE") if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index ddaed8b2b11..8f4c386fb59 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -3180,84 +3180,143 @@ test_busy_pane_repeated_escalation_reaches_demand_deep_inspection() { # A worker driving no-mistakes holds ONE turn open for the whole validation by # its generated Definition of done (bin/fm-dod-lib.sh), and a run chains fix # rounds well past BUSY_TURN_MAX_SECS, so a healthy validating crew crosses the -# completed-turn bound as a matter of course. The bound absorbs it on the -# attributed run-step alone - never on the busy pane, which cannot be its own -# bound - so only a run that is genuinely live silences the wedge escalator. -test_busy_pane_with_live_run_step_is_absorbed_past_turn_age_bound() { - local dir state fakebin out capture_file window key sig pid - dir=$(make_case busy-live-run-step); state="$dir/state"; fakebin="$dir/fakebin" +# completed-turn bound as a matter of course. The bound defers that escalation - +# it does not cancel it - and only on POSITIVE proof of both facts: an attributed +# working run-step AND a daemon a bounded probe proves up. The proof runs in the +# at-threshold branch only, so it costs at most one pair of bounded no-mistakes +# calls per window per FM_STALE_ESCALATE_SECS, never one per poll. +test_busy_pane_with_live_validation_defers_the_wedge_escalation() { + local dir state fakebin out capture_file window key pane_hash sig pid wt back + dir=$(make_case busy-live-validation); state="$dir/state"; fakebin="$dir/fakebin" out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-validating" + wt="$dir/wt"; mkdir -p "$wt" printf 'Working... (4210.6s)' > "$capture_file" - printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/validating.meta" + printf 'window=%s\nkind=ship\nharness=pi\nworktree=%s\n' "$window" "$wt" > "$state/validating.meta" record_pi_busy "$state" validating printf 'working: validating\n' > "$state/validating.status" sig=$(seen_sig "$state/validating.status"); printf '%s' "$sig" > "$state/.seen-validating_status" key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working... (4210.6s)") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" touch -t 200001010000 "$state/validating.turn-ended" prime_turnend_seen "$state/validating.turn-ended" + # The bound crossed long ago and the idle window opened 500s ago, so the very + # first poll lands straight on the at-threshold branch that consults the proof. + back=$(( $(date +%s) - 500 )) + echo "$back" > "$state/.stale-since-$key" + set_mtime "$back" "$state/.stale-since-$key" PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ - FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_FAKE_NM_DAEMON=up \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ + FM_PAUSE_RESURFACE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! if ! wait_poll_cycle "$state" "$pid"; then - reap "$pid"; fail "a busy pane with a live attributed run-step was wedge-escalated: $(cat "$out")" + reap "$pid"; fail "a busy pane with a proven-live validation was wedge-escalated: $(cat "$out")" fi - [ ! -s "$out" ] || fail "a busy pane with a live attributed run-step printed a wake reason" - [ ! -e "$state/.stale-since-$key" ] || fail "a live attributed run-step still started a wedge timer" - [ ! -e "$state/.wedge-escalations-$key" ] || fail "a live attributed run-step still counted a wedge escalation" + [ ! -s "$out" ] || { reap "$pid"; fail "a live-validation deferral printed a wake reason: $(cat "$out")"; } + [ ! -s "$state/.wake-queue" ] || { reap "$pid"; fail "a live-validation deferral enqueued a wake"; } + [ -e "$state/.nmrun-since-$key" ] || { reap "$pid"; fail "the live-validation deferral chain marker was not recorded"; } + [ ! -e "$state/.wedge-escalations-$key" ] || { reap "$pid"; fail "a live-validation deferral advanced the wedge escalation counter"; } + [ "$(cat "$state/.stale-since-$key" 2>/dev/null || echo 0)" -gt "$back" ] \ + || { reap "$pid"; fail "a live-validation deferral did not restart the idle timer, so the next window cannot re-prove the run"; } reap "$pid" - pass "a busy worker whose attributed no-mistakes run is still live is absorbed past the completed-turn bound" + pass "a busy worker whose no-mistakes validation is proven live defers the wedge escalation instead of firing it" } -# The complement of the absorb above, on the same fixture shape: run state that -# is no longer live - a terminal or unattributed record, or a daemon an explicit -# probe proves down (both reported by fm-crew-state.sh as something other than -# working/run-step) - is exactly the case the bound exists for, so it must still -# reach the wedge escalation with its stale reason and escalation counter. -test_busy_pane_with_dead_run_state_still_escalates_past_turn_age_bound() { - local dir state fakebin out capture_file window key sig pid - dir=$(make_case busy-dead-run-state); state="$dir/state"; fakebin="$dir/fakebin" - out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-dead-daemon" +# The regression the deferral must NOT swallow: after the shared daemon exits, the +# run record it left behind is never advanced, so `axi status` keeps reporting +# running/fixing and fm-crew-state.sh keeps reporting a working run-step - exactly +# the stale-record hazard rule 7 of the generated brief makes crews check for. The +# record alone therefore proves nothing; without the daemon probe answering up, +# the pane must still reach the unchanged escalation ladder. +test_busy_pane_with_dead_daemon_still_escalates_past_turn_age_bound() { + local dir state fakebin out drain_out capture_file window key pane_hash sig pid wt back + dir=$(make_case busy-dead-daemon); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; drain_out="$dir/drain.out"; capture_file="$dir/pane.txt" + window="test:fm-dead-daemon"; wt="$dir/wt"; mkdir -p "$wt" printf 'Working... (4210.6s)' > "$capture_file" - printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/dead-daemon.meta" + printf 'window=%s\nkind=ship\nharness=pi\nworktree=%s\n' "$window" "$wt" > "$state/dead-daemon.meta" record_pi_busy "$state" dead-daemon printf 'working: validating\n' > "$state/dead-daemon.status" sig=$(seen_sig "$state/dead-daemon.status"); printf '%s' "$sig" > "$state/.seen-dead-daemon_status" key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working... (4210.6s)") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" touch -t 200001010000 "$state/dead-daemon.turn-ended" prime_turnend_seen "$state/dead-daemon.turn-ended" + back=$(( $(date +%s) - 500 )) + echo "$back" > "$state/.stale-since-$key" + set_mtime "$back" "$state/.stale-since-$key" - # Phase A: the stale run record cannot vouch for the pane, so the bound starts - # the ordinary wedge timer exactly as it does with no run at all. + # Same still-running record as the deferral above; only the daemon probe differs. PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ - FM_FAKE_CREW_STATE='state: unknown · source: none · no-mistakes daemon unreachable; last ledger record failed - unverified' \ - FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ + FM_FAKE_NM_DAEMON=down \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ + FM_PAUSE_RESURFACE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! - if ! wait_poll_cycle "$state" "$pid"; then - reap "$pid"; fail "a busy pane with dead run state escalated before the wedge threshold: $(cat "$out")" - fi - [ -s "$state/.stale-since-$key" ] || fail "a busy pane with dead run state did not start a wedge timer" - reap "$pid" - ack_stopped_cycle "$state" || fail "could not acknowledge the intentional dead-run-state phase-A stop" + wait_for_exit "$pid" 100 || fail "a stale running record after a dead daemon did not wedge-escalate: $(cat "$out")" + grep -F "stale: $window" "$out" >/dev/null || fail "the dead-daemon escalation did not print the stale wake" + grep -F "possible wedge" "$out" >/dev/null || fail "the dead-daemon escalation did not flag a possible wedge" + [ "$(cat "$state/.wedge-escalations-$key" 2>/dev/null || true)" = 1 ] || fail "the dead-daemon escalation was not counted" + [ ! -e "$state/.nmrun-since-$key" ] || fail "a dead daemon still opened a live-validation deferral chain" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || fail "drain after the dead-daemon escalation failed" + grep "$(printf '\tstale\t')" "$drain_out" | grep -F "$window" >/dev/null || fail "the dead-daemon escalation was not queued" + pass "a busy worker whose daemon is down still escalates past the completed-turn bound, however live its run record reads" +} + +# A deferral is not silence here either: a validation can hold a live record while +# making no progress, so the whole live-run deferral chain ages and re-surfaces +# once per PAUSE_RESURFACE_SECS - the same bounded cadence a declared pause and a +# write deferral use - labeled as a recheck rather than a wedge. +test_live_validation_deferral_resurfaces_on_the_bounded_cadence() { + local dir state fakebin out drain_out capture_file window key pane_hash sig pid wt back + dir=$(make_case busy-live-validation-resurface); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; drain_out="$dir/drain.out"; capture_file="$dir/pane.txt" + window="test:fm-long-validation"; wt="$dir/wt"; mkdir -p "$wt" + printf 'Working... (9000.2s)' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\nworktree=%s\n' "$window" "$wt" > "$state/long-validation.meta" + record_pi_busy "$state" long-validation + printf 'working: validating\n' > "$state/long-validation.status" + sig=$(seen_sig "$state/long-validation.status"); printf '%s' "$sig" > "$state/.seen-long-validation_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working... (9000.2s)") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + touch -t 200001010000 "$state/long-validation.turn-ended" + prime_turnend_seen "$state/long-validation.turn-ended" + back=$(( $(date +%s) - 500 )) + echo "$back" > "$state/.stale-since-$key" + set_mtime "$back" "$state/.stale-since-$key" + # This pane has been deferring on the live run for 500s already. + : > "$state/.nmrun-since-$key" + set_mtime "$back" "$state/.nmrun-since-$key" - # Phase B: past the escalation threshold it wedge-escalates for human inspection. - echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" - : > "$out" PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ - FM_FAKE_CREW_STATE='state: unknown · source: none · no-mistakes daemon unreachable; last ledger record failed - unverified' \ - FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ + FM_FAKE_NM_DAEMON=up \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ + FM_PAUSE_RESURFACE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! - wait_for_exit "$pid" 100 || fail "a busy pane with dead run state did not wedge-escalate past the turn-age bound" - grep -F "stale: $window" "$out" >/dev/null || fail "dead run state escalation did not print the stale wake" - grep -F "possible wedge" "$out" >/dev/null || fail "dead run state escalation did not flag a possible wedge" - pass "a busy worker whose no-mistakes run state is stale or daemon-down still escalates past the completed-turn bound" + wait_for_exit "$pid" 100 || fail "a long live-validation deferral never re-surfaced on the bounded cadence" + grep -F "stale: $window" "$out" >/dev/null || fail "the live-validation recheck did not print a stale wake" + grep -F "no-mistakes validation proven live" "$out" >/dev/null || fail "the live-validation recheck was not labeled as such" + grep -F "possible wedge" "$out" >/dev/null && fail "a live-validation recheck was mislabeled a possible wedge" + [ -e "$state/.nmrun-resurfaced-$key" ] || fail "the live-validation re-surface throttle marker was not recorded" + [ ! -e "$state/.wedge-escalations-$key" ] || fail "a live-validation recheck advanced the wedge escalation counter" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || fail "drain after the live-validation recheck failed" + grep "$(printf '\tstale\t')" "$drain_out" | grep -F "$window" >/dev/null || fail "the live-validation recheck was not queued" + pass "a live-validation deferral re-surfaces once on the bounded pause cadence, so a run that stops advancing cannot stay invisible" } # --- declared pause + busy pane: the busy-turn bound must honor the declaration @@ -4516,8 +4575,9 @@ test_busy_pane_stable_hash_escalates_past_turn_age_bound test_busy_pane_changing_hash_escalates_past_turn_age_bound test_busy_pane_turn_end_touch_resets_age test_busy_pane_repeated_escalation_reaches_demand_deep_inspection -test_busy_pane_with_live_run_step_is_absorbed_past_turn_age_bound -test_busy_pane_with_dead_run_state_still_escalates_past_turn_age_bound +test_busy_pane_with_live_validation_defers_the_wedge_escalation +test_busy_pane_with_dead_daemon_still_escalates_past_turn_age_bound +test_live_validation_deferral_resurfaces_on_the_bounded_cadence test_busy_pane_default_turn_age_bound_is_3600s test_busy_declared_pause_is_rechecked_not_wedge_escalated test_afk_busy_declared_pause_hands_off_plain_stale diff --git a/tests/wake-helpers.sh b/tests/wake-helpers.sh index da83bb3dc91..d9a988b2d38 100644 --- a/tests/wake-helpers.sh +++ b/tests/wake-helpers.sh @@ -101,6 +101,7 @@ exit 1 SH chmod +x "$fakebin/tmux" make_fake_crew_state "$fakebin" >/dev/null + make_fake_no_mistakes "$fakebin" >/dev/null printf '%s\n' "$dir" } @@ -129,6 +130,29 @@ SH printf '%s\n' "$fakebin/fm-crew-state.sh" } +# Install a hermetic fake `no-mistakes` into and echo its path, so no +# test ever probes the machine's real shared daemon. Only `daemon status` is +# exercised (the liveness probe behind the wedge deferral for a live validation); +# FM_FAKE_NM_DAEMON=up makes the probe prove the daemon up, and anything else - +# including the default - leaves it unproven, the fail-closed answer a test that +# forgets to set one should get. +make_fake_no_mistakes() { # + local fakebin=$1 + cat > "$fakebin/no-mistakes" <<'SH' +#!/usr/bin/env bash +set -u +if [ "${1:-}" = daemon ] && [ "${2:-}" = status ]; then + case "${FM_FAKE_NM_DAEMON:-down}" in + up) echo "daemon: running"; exit 0 ;; + *) echo "daemon: not running" >&2; exit 1 ;; + esac +fi +exit 1 +SH + chmod +x "$fakebin/no-mistakes" + printf '%s\n' "$fakebin/no-mistakes" +} + # Prime 's .seen-* marker to its CURRENT signature through the production # signature owner (bin/fm-wake-lib.sh), so a test can declare "everything in # this file was already surfaced or deliberately absorbed" before exercising From 6900fd55c1f7d20ce4c935cf7c9a58ab4761ee86 Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 15:41:15 +0200 Subject: [PATCH 05/11] no-mistakes(review): Register nmrun markers and watcher suite selection --- AGENTS.md | 2 +- bin/fm-supervise-daemon.sh | 3 ++- bin/fm-test-run.sh | 8 +++++--- bin/fm-watch.sh | 10 +++++----- 4 files changed, 13 insertions(+), 10 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index ca09ab7a4cf..5285147c645 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -140,7 +140,7 @@ state/ runtime records and signals; gitignored .watch.lock .wake-queue.lock watcher singleton and queue serialization locks .claude-autoarm.lock .claude-autoarm-epoch .claude-autoarm-failure-notified .claude-autoarm-failure-alarmed .turnend-claude-blocks .turnend-claude-blocks.lock Claude Stop auto-arm single-flight, epoch, failure-episode, attended-alarm, guard-budget, and budget-lock records; never touch .cursor-park-owner .cursor-park-owner.lock .turnend-cursor-blocks Cursor stop-hook owner record, publication and commit lock, and bounded repair-nag budget; never touch - .hash-* .count-* .stale-* .stale-since-* .churn-since-* .paused-* .wedge-escalations-* .writing-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch + .hash-* .count-* .stale-* .stale-since-* .churn-since-* .paused-* .wedge-escalations-* .writing-* .nmrun-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch .watch-triage.log watcher's absorbed-wake debug log (size-capped); never relied on, safe to delete .last-watcher-beat watcher liveness beacon, touched every poll (including while absorbing benign wakes); guard scripts read it .subsuper-* .supervise-daemon.* sub-supervisor internals; never touch diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index 91174bc5baf..9c939fac741 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -512,7 +512,8 @@ clear_pause_tracking() { # rm -f "$state/.subsuper-paused-$key" "$state/.subsuper-stale-$key" \ "$state/.paused-$watcher_key" "$state/.paused-rechecked-$watcher_key" "$state/.paused-resurfaced-$watcher_key" \ "$state/.stale-$watcher_key" "$state/.stale-since-$watcher_key" "$state/.wedge-escalations-$watcher_key" \ - "$state/.writing-since-$watcher_key" "$state/.writing-resurfaced-$watcher_key" + "$state/.writing-since-$watcher_key" "$state/.writing-resurfaced-$watcher_key" \ + "$state/.nmrun-since-$watcher_key" "$state/.nmrun-resurfaced-$watcher_key" } reconcile_pause_tracking() { # diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 762272b948f..b3a4f87beee 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -1351,11 +1351,13 @@ families_for_changed_path() { printf '%s\n' pr-forge ;; bin/fm-nm-run-lib.sh) - # Shared no-mistakes run-attribution primitives, sourced by both - # bin/fm-crew-state.sh (pure-contract-unit) and bin/fm-teardown.sh's - # pre-teardown run abort (pr-forge). + # Shared no-mistakes run-attribution and daemon-liveness primitives, sourced + # by bin/fm-crew-state.sh (pure-contract-unit), bin/fm-teardown.sh's + # pre-teardown run abort (pr-forge), and bin/fm-classify-lib.sh, whose + # live-validation wedge deferral is exercised by the watcher suite. printf '%s\n' pure-contract-unit printf '%s\n' pr-forge + printf '%s\n' watcher-wake-lock ;; bin/fm-control-lib.sh) printf '%s\n' backend-dispatch diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 9b69ca5cd7c..ccc49ceb219 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -323,11 +323,11 @@ window_label() { # The ONE derivation of a window's per-window marker key: `:`, `/` and `.` become # `_` so a window name is usable as a filename suffix. Every per-window file the # watcher keeps is named by it (.hash-, .count-, .stale-, .stale-since-, -# .wedge-escalations-, .paused-*, .writing-*), and live homes hold those markers on -# disk under the current format, so the format lives here alone: a second copy is -# how a future change to it silently orphans a window's markers instead of clearing -# them. The helpers below take the derived key rather than re-deriving it, so one -# poll of one window derives it once. +# .wedge-escalations-, .paused-*, .writing-*, .nmrun-*), and live homes hold those +# markers on disk under the current format, so the format lives here alone: a +# second copy is how a future change to it silently orphans a window's markers +# instead of clearing them. The helpers below take the derived key rather than +# re-deriving it, so one poll of one window derives it once. window_key() { # local key=${1//:/_} key=${key//\//_} From 76adfb91d6990e4cbb07d33c1ec9c77aa5d9a49f Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 15:55:40 +0200 Subject: [PATCH 06/11] no-mistakes(review): Restore continuation rule for start and reattach --- bin/fm-dod-lib.sh | 1 + tests/fm-brief.test.sh | 2 ++ 2 files changed, 3 insertions(+) diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index 6dff89705d3..b3c304b3dda 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -242,6 +242,7 @@ Where a harness's own command limit is not established, assume it bounds command A killed or timed-out call is never evidence the daemon died: the daemon accepts your response immediately and runs the round in the background, so the call was only ever waiting for a read while the run kept working. Reattach and keep going rather than reporting the pipeline blocked; rule 7 owns the checks that decide when a pipeline block is real. After every \`no-mistakes axi respond\`, continue in the same turn with bounded calls to the structured \`no-mistakes axi status\` interface until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block. +An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run. Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block. Two firstmate-specific rules layer on top of that guidance: diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 17f4fa410f9..9567f7fd292 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -386,6 +386,8 @@ test_active_no_mistakes_validation_cannot_be_deferred() { "no-mistakes brief did not require bounded structured status polling" assert_grep "until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block" "$brief" \ "no-mistakes brief did not define the only status-polling stop conditions" + assert_grep "An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run." "$brief" \ + "no-mistakes brief did not extend the continuation rule past a gate response to starting and reattaching" assert_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block." "$brief" \ "no-mistakes brief still permits deferring an active validation run" pass "fm-brief.sh: active no-mistakes validation continues in the same turn through the next real transition" From 4e8a1fafd6d584a1adf1cac24868efc08b2ea530 Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 16:38:05 +0200 Subject: [PATCH 07/11] no-mistakes(review): Defer busy-turn wedge on this run's own activity --- AGENTS.md | 2 +- bin/fm-classify-lib.sh | 73 +++++++---------- bin/fm-crew-state.sh | 20 ++++- bin/fm-nm-run-lib.sh | 9 --- bin/fm-supervise-daemon.sh | 3 +- bin/fm-test-run.sh | 8 +- bin/fm-watch.sh | 116 +++++++++++---------------- tests/fm-crew-state.test.sh | 36 +++++++++ tests/fm-watch-triage.test.sh | 147 +++++++++------------------------- tests/wake-helpers.sh | 24 ------ 10 files changed, 175 insertions(+), 263 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 5285147c645..ca09ab7a4cf 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -140,7 +140,7 @@ state/ runtime records and signals; gitignored .watch.lock .wake-queue.lock watcher singleton and queue serialization locks .claude-autoarm.lock .claude-autoarm-epoch .claude-autoarm-failure-notified .claude-autoarm-failure-alarmed .turnend-claude-blocks .turnend-claude-blocks.lock Claude Stop auto-arm single-flight, epoch, failure-episode, attended-alarm, guard-budget, and budget-lock records; never touch .cursor-park-owner .cursor-park-owner.lock .turnend-cursor-blocks Cursor stop-hook owner record, publication and commit lock, and bounded repair-nag budget; never touch - .hash-* .count-* .stale-* .stale-since-* .churn-since-* .paused-* .wedge-escalations-* .writing-* .nmrun-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch + .hash-* .count-* .stale-* .stale-since-* .churn-since-* .paused-* .wedge-escalations-* .writing-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch .watch-triage.log watcher's absorbed-wake debug log (size-capped); never relied on, safe to delete .last-watcher-beat watcher liveness beacon, touched every poll (including while absorbing benign wakes); guard scripts read it .subsuper-* .supervise-daemon.* sub-supervisor internals; never touch diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index a3508696307..246df9e2b46 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -61,9 +61,6 @@ case $- in *u*) _fm_classify_nounset=on ;; *) _fm_classify_nounset=off ;; esac # shellcheck source=bin/fm-timeout-lib.sh # shellcheck disable=SC1091 . "$_FM_CLASSIFY_LIB_DIR/fm-timeout-lib.sh" -# shellcheck source=bin/fm-nm-run-lib.sh -# shellcheck disable=SC1091 -. "$_FM_CLASSIFY_LIB_DIR/fm-nm-run-lib.sh" [ "$_fm_classify_nounset" = on ] || set +u unset _fm_classify_nounset @@ -1747,22 +1744,6 @@ status_span_has_actionable() { # status_span_first_actionable_record "$1" "${2:-0}" > /dev/null } -# Read bin/fm-crew-state.sh's one authoritative current-state line -# ("state: · source: · ") and print it as " ", -# the two fields every absorb decision is made from. Fails (prints nothing) when -# no id is given or the read is missing/unparseable, so a caller cannot mistake an -# unreadable crew for a classified one. Not a pure read: see crew_absorb_class. -# FM_CREW_STATE_BIN lets tests stub the verdict. -crew_state_verdict() { # - local id=$1 line state src= - [ -n "$id" ] || return 1 - line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || true - case "$line" in state:*) ;; *) return 1 ;; esac - state=${line#state: }; state=${state%% *} - case "$line" in *"source: "*) src=${line#*source: }; src=${src%% *} ;; esac - printf '%s %s' "$state" "$src" -} - # Classify WHY an idle/stale crew MIGHT be safely absorbed instead of surfaced, # from bin/fm-crew-state.sh's one authoritative current-state line # ("state: · source: · "). Prints exactly one token: @@ -1780,11 +1761,14 @@ crew_state_verdict() { # # run it only on no-verb signal and first-sighting stale paths, never every wake. # FM_CREW_STATE_BIN lets tests stub the verdict. crew_absorb_class() { # - local verdict state src - verdict=$(crew_state_verdict "$1") || { printf 'none'; return; } - state=${verdict%% *}; src=${verdict##* } + local id=$1 line state src + [ -n "$id" ] || { printf 'none'; return; } + line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || true + case "$line" in state:*) ;; *) printf 'none'; return ;; esac + state=${line#state: }; state=${state%% *} if [ "$state" = paused ]; then printf 'paused'; return; fi if [ "$state" = working ]; then + src=${line#*source: }; src=${src%% *} case "$src" in run-step|pane) printf 'working'; return ;; esac fi printf 'none' @@ -1811,28 +1795,33 @@ crew_is_paused() { # [ "$(crew_absorb_class "$1")" = paused ] } -# 0 only on POSITIVE proof that crew 's no-mistakes validation is running -# right now, from two facts that must BOTH hold: -# - fm-crew-state.sh attributes a working run-step to this crew (branch AND code -# identity, or pipeline-owned custody), which rules out a terminal run and a -# record that no longer belongs to this worktree; -# - fm_nm_daemon_is_alive proves the shared daemon is up. -# The second is not redundant: a run record left at running/fixing is never -# advanced after the daemon exits, so the record alone reports a live run for a -# validation nothing is executing - the same stale-record hazard rule 7 of the -# generated brief makes crews check before appending `blocked:`. Every failure is -# a negative answer, so a missing worktree, an unreadable verdict, and a probe -# that times out all read as NOT live. +# The note bin/fm-crew-state.sh appends to an active run-step's detail when the +# pipeline's own recency verdict says that step is still producing activity. One +# definition, written by fm-crew-state.sh and matched by the predicate below, so +# the emitted line and the classifier reading it cannot drift apart. +FM_CREW_STATE_ACTIVITY_RECENT='run activity recent' + +# 0 only on POSITIVE proof that crew 's OWN attributed no-mistakes run is +# still doing work: fm-crew-state.sh reports a working run-step for THIS crew and +# marks its active step's activity recent, which is the pipeline's own recency +# verdict (`axi status` prefixes last_activity with `quiet` once nothing has +# arrived), never a second threshold invented here and never the liveness of the +# shared daemon, which any other crew's run keeps up. That distinction is the +# whole point: a record left at running/fixing after a drive call was killed, or +# after the daemon exited under it, reports a working run-step while nothing +# executes it, and must NOT read as work in progress. # The busy-pane half of crew_absorb_class's `working` is deliberately excluded: a # caller that already holds a busy verdict cannot let that pane vouch for itself. -# Two bounded subprocesses per call, so callers must run it at most once per -# STALE_ESCALATE_SECS - never per poll (see crew_absorb_class). -crew_nm_run_is_provably_live() { # - local id=$1 state=$2 wt - [ "$(crew_state_verdict "$id")" = "working run-step" ] || return 1 - wt=$(grep '^worktree=' "$state/$id.meta" 2>/dev/null | tail -1 | cut -d= -f2- || true) - [ -n "$wt" ] && [ -d "$wt" ] || return 1 - fm_nm_daemon_is_alive "$wt" "${FM_CREW_STATE_NM_TIMEOUT:-10}" +# Not a pure read (see crew_absorb_class), so callers run it at most once per +# STALE_ESCALATE_SECS - never per poll. +crew_nm_run_activity_is_recent() { # + local id=$1 line + [ -n "$id" ] || return 1 + line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || true + case "$line" in + "state: working"*"source: run-step"*"$FM_CREW_STATE_ACTIVITY_RECENT"*) return 0 ;; + esac + return 1 } # Directories excluded from the worktree write probe below, and the depth it walks. diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index fc54687952b..fc4924f06a7 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -453,11 +453,15 @@ nm_reclassify_failed_run_as_held_green() { return 0 } -# 0 when an explicit probe proves the shared daemon down - the negation of -# fm_nm_daemon_is_alive in bin/fm-nm-run-lib.sh, which owns that one probe for -# every caller that needs daemon liveness. +# 0 when an explicit probe proves the shared daemon down: `no-mistakes daemon +# status` is the canonical down-probe (the same one fm-brief.sh hands crews +# before a blocked append) and exits non-zero when the daemon is not running. +# Bounded like every other CLI call; a probe that fails for any reason - +# refused socket, timeout, non-zero answer - means the daemon is not provably +# up, which is the only fact the coarse fallback needs. nm_daemon_probe_down() { - ! fm_nm_daemon_is_alive "$WT" "$NM_TIMEOUT" + fm_nm_run_checked "$WT" "$NM_TIMEOUT" daemon status >/dev/null || return 0 + return 1 } nm_ci_step_status() { @@ -739,6 +743,14 @@ if [ "$HAVE_RUN" = 1 ]; then ;; esac + # Positive recency, for supervisors that must tell an advancing run from a + # record nothing is executing: the client's own `quiet` prefix is the verdict + # (nm_run_activity_is_recent), so the note appears only while an active step + # keeps reporting, and never for a coarse row with no steps table to read. + if [ "$RUN_STATE" = working ] && nm_run_activity_is_recent; then + RUN_DETAIL="$RUN_DETAIL${SEP}$FM_CREW_STATE_ACTIVITY_RECENT" + fi + emit "$RUN_STATE" run-step "$RUN_DETAIL" fi diff --git a/bin/fm-nm-run-lib.sh b/bin/fm-nm-run-lib.sh index 328ba28ef42..9a04004e391 100644 --- a/bin/fm-nm-run-lib.sh +++ b/bin/fm-nm-run-lib.sh @@ -38,15 +38,6 @@ fm_nm_run() { # fm_nm_run_checked "$@" || true } -# 0 only when a bounded `no-mistakes daemon status` PROVES the shared daemon up. -# The canonical liveness probe (the same one fm-brief.sh hands crews before a -# blocked append), stated positively: any failure - refused socket, timeout, -# missing binary, non-zero answer - means the daemon is not provably up, so every -# caller that needs daemon liveness fails closed on the same one fact. -fm_nm_daemon_is_alive() { # - fm_nm_run_checked "$1" "$2" daemon status >/dev/null -} - fm_nm_trim() { local s=${1:-} s="${s#"${s%%[![:space:]]*}"}" diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index 9c939fac741..91174bc5baf 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -512,8 +512,7 @@ clear_pause_tracking() { # rm -f "$state/.subsuper-paused-$key" "$state/.subsuper-stale-$key" \ "$state/.paused-$watcher_key" "$state/.paused-rechecked-$watcher_key" "$state/.paused-resurfaced-$watcher_key" \ "$state/.stale-$watcher_key" "$state/.stale-since-$watcher_key" "$state/.wedge-escalations-$watcher_key" \ - "$state/.writing-since-$watcher_key" "$state/.writing-resurfaced-$watcher_key" \ - "$state/.nmrun-since-$watcher_key" "$state/.nmrun-resurfaced-$watcher_key" + "$state/.writing-since-$watcher_key" "$state/.writing-resurfaced-$watcher_key" } reconcile_pause_tracking() { # diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index b3a4f87beee..762272b948f 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -1351,13 +1351,11 @@ families_for_changed_path() { printf '%s\n' pr-forge ;; bin/fm-nm-run-lib.sh) - # Shared no-mistakes run-attribution and daemon-liveness primitives, sourced - # by bin/fm-crew-state.sh (pure-contract-unit), bin/fm-teardown.sh's - # pre-teardown run abort (pr-forge), and bin/fm-classify-lib.sh, whose - # live-validation wedge deferral is exercised by the watcher suite. + # Shared no-mistakes run-attribution primitives, sourced by both + # bin/fm-crew-state.sh (pure-contract-unit) and bin/fm-teardown.sh's + # pre-teardown run abort (pr-forge). printf '%s\n' pure-contract-unit printf '%s\n' pr-forge - printf '%s\n' watcher-wake-lock ;; bin/fm-control-lib.sh) printf '%s\n' backend-dispatch diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index ccc49ceb219..1932c6de0eb 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -214,10 +214,10 @@ STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provabl # non-busy stale - so it escalates via the existing stale reason, escalation # counter, and demand-deep-inspection marker for human inspection only, never an # automatic interrupt, signal, or restart - unless the crew declared the wait -# itself, which takes the long pause cadence instead, or a no-mistakes validation -# is proven live at the escalation threshold, which is deferred onto that same -# bounded cadence for as long as the proof keeps holding. A completed turn touches -# turn-ended and resets the age. Set generously above any legitimate interval +# itself, which takes the long pause cadence instead, or the crew's own +# no-mistakes run reports recent activity at the escalation threshold, which +# defers that one escalation and must prove itself again for the next. +# A completed turn touches turn-ended and resets the age. Set generously above any legitimate interval # between completed turns, including long tool calls, builds, or test runs. BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # A local secondmate's foreign queue is checked on every poll, but only after this @@ -323,11 +323,11 @@ window_label() { # The ONE derivation of a window's per-window marker key: `:`, `/` and `.` become # `_` so a window name is usable as a filename suffix. Every per-window file the # watcher keeps is named by it (.hash-, .count-, .stale-, .stale-since-, -# .wedge-escalations-, .paused-*, .writing-*, .nmrun-*), and live homes hold those -# markers on disk under the current format, so the format lives here alone: a -# second copy is how a future change to it silently orphans a window's markers -# instead of clearing them. The helpers below take the derived key rather than -# re-deriving it, so one poll of one window derives it once. +# .wedge-escalations-, .paused-*, .writing-*), and live homes hold those markers on +# disk under the current format, so the format lives here alone: a second copy is +# how a future change to it silently orphans a window's markers instead of clearing +# them. The helpers below take the derived key rather than re-deriving it, so one +# poll of one window derives it once. window_key() { # local key=${1//:/_} key=${key//\//_} @@ -795,9 +795,8 @@ FM_WEDGE_DEMAND_INSPECT_COUNT=${FM_WEDGE_DEMAND_INSPECT_COUNT:-3} # once past PAUSE_RESURFACE_SECS the pane wakes once per window rather than every # poll. An optional binds that cadence to its current declaration; callers # without a scoped declaration keep the timestamp body. Shared by the -# declared-pause absorb and both wedge deferrals (worktree-write and live -# no-mistakes run) so their cadences cannot drift apart; each caller owns its own -# marker and reason. +# declared-pause absorb and the worktree-write deferral so the two cadences cannot +# drift apart; each caller owns its own marker and reason. # Returns without waking while either the absorb or the throttle is inside the # window; wake() itself exits the cycle, exactly as it does inline. resurface_absorbed() { # [scope] @@ -838,39 +837,12 @@ wedge_defer_writing() { # triage_log "absorbed $label (worktree written since the idle window opened, idle ${age}s): $win" } -# Defer ONE wedge escalation for a pane whose no-mistakes validation is provably -# live (crew_nm_run_is_provably_live: an attributed working run-step AND a daemon -# proven up). This is the busy-turn bound's second real wait: the crew has not -# completed a turn because the generated Definition of done forbids ending the -# turn while its validation is active, so the pipeline - not the pane - is the -# evidence. Deliberately the same DEFERRAL shape as the worktree-write one, never -# a cancellation: the idle timer restarts so the next window re-proves both facts -# from scratch, and a .nmrun-since- chain ages the whole deferral so the pane -# still re-surfaces once every PAUSE_RESURFACE_SECS through resurface_absorbed. A -# run that stops advancing therefore cannot stay invisible, and the moment the -# record goes terminal, loses attribution, or the daemon dies, the next threshold -# falls straight through to the unchanged escalation ladder. The escalation -# counter is left alone for the same reason wedge_defer_writing leaves it. -wedge_defer_live_run() { # - local win=$1 since_file=$2 label=$3 age=$4 key rsf rage - key=$(window_key "$win") - rsf="$STATE/.nmrun-since-$key" - [ -e "$rsf" ] || date +%s > "$rsf" - rage=$(age_of "$rsf") - date +%s > "$since_file" - resurface_absorbed "$win" "$STATE/.nmrun-resurfaced-$key" "$rage" \ - "stale: $win (idle ${age}s, no-mistakes validation proven live for ${rage}s, rechecked on a long cadence not a wedge; confirm the run is still advancing)" - triage_log "absorbed $label (attributed no-mistakes run proven live, idle ${age}s): $win" -} - -# Drop a window's deferral chains - worktree-write and live-run alike - wherever -# its stale bookkeeping resets, so the bounded re-surface cadence is measured from -# the CURRENT quiet stretch and a long-finished chain cannot make the next -# deferral resurface immediately. -clear_defer_tracking() { # +# Drop a window's write-deferral chain wherever its stale bookkeeping resets, so +# the bounded re-surface cadence is measured from the CURRENT quiet stretch and a +# long-finished one cannot make the next deferral resurface immediately. +clear_write_tracking() { # local key=$1 - rm -f "$STATE/.writing-since-$key" "$STATE/.writing-resurfaced-$key" \ - "$STATE/.nmrun-since-$key" "$STATE/.nmrun-resurfaced-$key" + rm -f "$STATE/.writing-since-$key" "$STATE/.writing-resurfaced-$key" } # Repeat-poll wedge-timer bookkeeping for an already-classified stale hash @@ -883,7 +855,14 @@ clear_defer_tracking() { # # line that an active run/busy pane outranked). # The worktree write probe runs ONLY here, inside the at-threshold branch that is # about to escalate: at most one bounded walk per window per STALE_ESCALATE_SECS, -# never per poll. +# never per poll. `defer-live-run` adds the one other evidence a pane can offer +# there, on the same terms and for the same reason: this crew's OWN no-mistakes +# run reporting recent activity defers the escalation once, restarting the idle +# timer so the next window must prove it again. The moment that run stops +# reporting recent activity - a hung step, a record left behind after the daemon +# exited, a run that is no longer this crew's - the threshold falls straight +# through to the unchanged escalation ladder and its demand-deep-inspection +# marker, so no deferral can repeat without fresh proof of work. wedge_timer_check() { # [defer-live-run] local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 live_run=${6-} since age n reason since=$(cat "$since_file" 2>/dev/null || true) @@ -891,7 +870,7 @@ wedge_timer_check() { # "$since_file" triage_log "absorbed $label timer reset: $win" ;; @@ -902,8 +881,9 @@ wedge_timer_check() { # "$since_file" + triage_log "absorbed $label (this crew's own no-mistakes run reports recent activity, idle ${age}s): $win" return 0 fi n=$(( $(cat "$escalation_file" 2>/dev/null || echo 0) + 1 )) @@ -914,7 +894,7 @@ wedge_timer_check() { # printf '%s' "$h" > "$STATE/.stale-$key" : > "$STATE/.paused-$key" rm -f "$STATE/.stale-since-$key" "$STATE/.wedge-escalations-$key" - clear_defer_tracking "$key" + clear_write_tracking "$key" statusf="$STATE/$task.status" mtime=$(stat_mtime "$statusf") case "$mtime" in ''|*[!0-9]*) mtime=$(date +%s) ;; esac @@ -981,16 +961,16 @@ handle_paused_stale() { # # A busy pane past BUSY_TURN_MAX_SECS is normally a wedge suspect because a hung # foreground call can hide behind a busy signature. A `paused:` declaration or # verified captain-held transfer instead identifies that live foreground call as -# the expected external wait. A live no-mistakes validation is the other real -# wait - a validating worker holds ONE turn open for the whole run by contract, -# so its completed-turn age is expected to cross the bound - but that evidence is -# NOT read here: it is handed to wedge_timer_check as `defer-live-run`, which -# consults it only in the branch that is about to escalate. That keeps the two -# bounded no-mistakes subprocesses to at most one pair per window per -# STALE_ESCALATE_SECS instead of one pair per poll, and keeps the outcome a -# bounded deferral rather than an unbounded silence. The caller has already confirmed liveness through -# the busy verdict, so this exception does not suppress undeclared wedges or -# alter the separate non-busy classification. handle_paused_stale keeps the +# the expected external wait. An advancing no-mistakes validation is the other +# real wait - a validating worker holds ONE turn open for the whole run by +# contract, so its completed-turn age is expected to cross the bound - but that +# evidence is NOT read here: it is handed to wedge_timer_check as +# `defer-live-run`, which consults it only in the branch that is about to +# escalate. That keeps the crew-state read to at most one per window per +# STALE_ESCALATE_SECS instead of one per poll, and keeps the outcome a single +# deferral that must be re-earned rather than a standing exemption. The caller +# has already confirmed liveness through the busy verdict, so this exception does +# not suppress undeclared wedges or alter the separate non-busy classification. handle_paused_stale keeps the # exception bounded by re-surfacing it once per PAUSE_RESURFACE_SECS. Away mode # remains daemon-owned and receives the undecorated wake identity for its own # classification, which is why the declaration is read before the afk branch @@ -1021,7 +1001,7 @@ busy_turn_bound_check() { # /dev/null || true)" != "$declared" ]; then fm_wake_append stale "$win" "stale: $win" || exit 1 @@ -1048,7 +1028,7 @@ clear_pause_state() { # # recheck, and re-surface throttle - can still reset the per-hash half alone. clear_stale_hash_tracking() { # local key=$1 - clear_defer_tracking "$key" + clear_write_tracking "$key" rm -f "$STATE/.stale-$key" "$STATE/.stale-since-$key" "$STATE/.wedge-escalations-$key" } @@ -1253,7 +1233,7 @@ surface_nonterminal_stale() { # fi printf '%s' "$h" > "$STATE/.stale-$key" rm -f "$STATE/.stale-since-$key" - clear_defer_tracking "$key" + clear_write_tracking "$key" if [ "$declared" -eq 0 ]; then : > "$STATE/.paused-$key" date +%s > "$STATE/.paused-rechecked-$key" @@ -2148,7 +2128,7 @@ EOF if crew_is_provably_working "$(window_to_task "$w" "$STATE")"; then printf '%s' "$h" > "$sf" date +%s > "$ssf" - clear_defer_tracking "$key" + clear_write_tracking "$key" triage_log "absorbed stale (provably working, overriding a stale captain-relevant status): $w" elif captain_call_stale_bound "$key" "$task"; then # The line is captain-relevant and stays so, but the backlog says @@ -2160,14 +2140,14 @@ EOF # here as it already was after a first terminal alarm. printf '%s' "$h" > "$sf" rm -f "$ssf" - clear_defer_tracking "$key" + clear_write_tracking "$key" triage_log "absorbed stale (open captain call already surfaced for this status): $w" else fm_wake_append stale "$w" "stale: $w" || exit 1 stale_wait_record "$key" printf '%s' "$h" > "$sf" rm -f "$ssf" - clear_defer_tracking "$key" + clear_write_tracking "$key" stale_status="$STATE/$(window_to_task "$w" "$STATE").status" stale_record=$(status_span_first_actionable_record "$stale_status" 0) case $? in @@ -2244,7 +2224,7 @@ EOF busy_turn_bound_check "$w" "$task" "$h" "$ssf" "$ewf" && paused_bound=0 else rm -f "$ssf" "$ewf" - clear_defer_tracking "$key" + clear_write_tracking "$key" fi # A busy pane normally means real work resumed, so stale pause bookkeeping # is cleared - but not in the same poll the declared-pause cadence just @@ -2262,7 +2242,7 @@ EOF busy_turn_bound_check "$w" "$task" "$h" "$ssf" "$ewf" && paused_bound=0 else rm -f "$ssf" "$ewf" - clear_defer_tracking "$key" + clear_write_tracking "$key" fi task=$(window_to_task "$w" "$STATE") if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 309a7008f8a..4620030e6bd 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -567,6 +567,41 @@ test_daemon_claim_over_live_run_reads_run_alive() { pass "daemon/timeout blocked claim over a live fixing run reads as run alive" } +# The emitted line is what supervisors classify from, so an ACTIVE run's own +# recency verdict has to reach it. `axi status` leaves last_activity unprefixed +# while a step keeps reporting and prefixes it with `quiet` once nothing arrives; +# only the first case earns the recency note, so a record still reading fixing +# while nothing executes it is distinguishable from a run doing work. +test_active_run_reports_its_own_activity_recency() { + reset_fakes + local d; d=$(new_case activity-recency) + make_repo_on_branch "$d/wt" fm/feat-ar + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-ar.meta" "window=fm:fm-feat-ar" "worktree=$d/wt" "kind=ship" + + FM_FAKE_AXI_STATUS="$(run_fixing_active_recent fm/feat-ar)" + local out; out=$(run_crew_state "$d" feat-ar) + assert_contains "$out" "state: working" "a reporting fixing run is working" + assert_contains "$out" "run activity recent" "a reporting active step earns the recency note" + + FM_FAKE_AXI_STATUS="$(run_fixing_active_quiet fm/feat-ar)" + out=$(run_crew_state "$d" feat-ar) + assert_contains "$out" "source: run-step" "a quiet fixing run is still run-step sourced" + assert_not_contains "$out" "run activity recent" \ + "a quiet active step must not read as a run doing work" + + # The coarse ledger fallback has no steps table to read, so it can never claim + # recency: absent positive evidence, the note stays off. + local short; short=$(git -C "$d/wt" rev-parse --short=7 HEAD) + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST=" running fm/feat-ar ${short} 2026-07-02 22:05" + out=$(run_crew_state "$d" feat-ar) + assert_contains "$out" "state: working" "the coarse row still attributes this branch's run" + assert_not_contains "$out" "run activity recent" \ + "the coarse runs-list fallback cannot claim activity recency" + pass "an active run-step reports the pipeline's own activity recency, and only on positive evidence" +} + # A genuine refused socket outranks the persisted fixing record, which can # survive after the daemon exits. test_socket_refusal_over_stale_fixing_run_reports_blocked() { @@ -2235,6 +2270,7 @@ test_active_run_is_authoritative test_stale_needs_decision_superseded test_stale_blocked_superseded test_daemon_claim_over_live_run_reads_run_alive +test_active_run_reports_its_own_activity_recency test_socket_refusal_over_stale_fixing_run_reports_blocked test_socket_refusal_over_terminal_run_reports_blocked test_ordinary_blocked_over_live_run_keeps_plain_superseded diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index 8f4c386fb59..76c74998388 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -3177,21 +3177,24 @@ test_busy_pane_repeated_escalation_reaches_demand_deep_inspection() { } # --- live no-mistakes run + busy pane: the run IS the declared wait ---------- -# A worker driving no-mistakes holds ONE turn open for the whole validation by -# its generated Definition of done (bin/fm-dod-lib.sh), and a run chains fix -# rounds well past BUSY_TURN_MAX_SECS, so a healthy validating crew crosses the -# completed-turn bound as a matter of course. The bound defers that escalation - -# it does not cancel it - and only on POSITIVE proof of both facts: an attributed -# working run-step AND a daemon a bounded probe proves up. The proof runs in the -# at-threshold branch only, so it costs at most one pair of bounded no-mistakes -# calls per window per FM_STALE_ESCALATE_SECS, never one per poll. -test_busy_pane_with_live_validation_defers_the_wedge_escalation() { - local dir state fakebin out capture_file window key pane_hash sig pid wt back - dir=$(make_case busy-live-validation); state="$dir/state"; fakebin="$dir/fakebin" - out="$dir/watch.out"; capture_file="$dir/pane.txt"; window="test:fm-validating" - wt="$dir/wt"; mkdir -p "$wt" +# A worker driving no-mistakes holds ONE turn open for the whole validation by its +# generated Definition of done (bin/fm-dod-lib.sh), and a run chains fix rounds +# well past BUSY_TURN_MAX_SECS, so a healthy validating crew crosses the +# completed-turn bound as a matter of course. Its escalation is DEFERRED, never +# cancelled, and only on the pipeline's own recency verdict for THIS crew's run: +# `axi status` prefixes an active step's last_activity with `quiet` once nothing +# is arriving, and fm-crew-state.sh only then withholds the recency note. Phases +# A and B pin both halves on one window - recent activity defers, the same pane +# with the activity gone escalates on the unchanged schedule - so a deferral can +# never become a standing exemption for a hung run or for a record left behind +# after the daemon exited under it. +test_busy_turn_bound_defers_only_while_the_run_reports_recent_activity() { + local dir state fakebin out drain_out capture_file window key pane_hash sig pid back + dir=$(make_case busy-run-activity); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; drain_out="$dir/drain.out"; capture_file="$dir/pane.txt" + window="test:fm-validating" printf 'Working... (4210.6s)' > "$capture_file" - printf 'window=%s\nkind=ship\nharness=pi\nworktree=%s\n' "$window" "$wt" > "$state/validating.meta" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/validating.meta" record_pi_busy "$state" validating printf 'working: validating\n' > "$state/validating.status" sig=$(seen_sig "$state/validating.status"); printf '%s' "$sig" > "$state/.seen-validating_status" @@ -3202,121 +3205,51 @@ test_busy_pane_with_live_validation_defers_the_wedge_escalation() { touch -t 200001010000 "$state/validating.turn-ended" prime_turnend_seen "$state/validating.turn-ended" # The bound crossed long ago and the idle window opened 500s ago, so the very - # first poll lands straight on the at-threshold branch that consults the proof. + # first poll lands straight on the at-threshold branch that reads the evidence. back=$(( $(date +%s) - 500 )) echo "$back" > "$state/.stale-since-$key" set_mtime "$back" "$state/.stale-since-$key" + # Phase A: this crew's own run reports recent activity. Deferred. PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ - FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ - FM_FAKE_NM_DAEMON=up \ + FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing) · run activity recent' \ FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ FM_PAUSE_RESURFACE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! if ! wait_poll_cycle "$state" "$pid"; then - reap "$pid"; fail "a busy pane with a proven-live validation was wedge-escalated: $(cat "$out")" + reap "$pid"; fail "a busy pane whose own run reports recent activity was wedge-escalated: $(cat "$out")" fi - [ ! -s "$out" ] || { reap "$pid"; fail "a live-validation deferral printed a wake reason: $(cat "$out")"; } - [ ! -s "$state/.wake-queue" ] || { reap "$pid"; fail "a live-validation deferral enqueued a wake"; } - [ -e "$state/.nmrun-since-$key" ] || { reap "$pid"; fail "the live-validation deferral chain marker was not recorded"; } - [ ! -e "$state/.wedge-escalations-$key" ] || { reap "$pid"; fail "a live-validation deferral advanced the wedge escalation counter"; } + [ ! -s "$out" ] || { reap "$pid"; fail "a recent-activity deferral printed a wake reason: $(cat "$out")"; } + [ ! -s "$state/.wake-queue" ] || { reap "$pid"; fail "a recent-activity deferral enqueued a wake"; } + [ ! -e "$state/.wedge-escalations-$key" ] || { reap "$pid"; fail "a recent-activity deferral advanced the wedge escalation counter"; } [ "$(cat "$state/.stale-since-$key" 2>/dev/null || echo 0)" -gt "$back" ] \ - || { reap "$pid"; fail "a live-validation deferral did not restart the idle timer, so the next window cannot re-prove the run"; } + || { reap "$pid"; fail "a recent-activity deferral did not restart the idle timer, so the next window cannot re-prove the run"; } reap "$pid" - pass "a busy worker whose no-mistakes validation is proven live defers the wedge escalation instead of firing it" -} + ack_stopped_cycle "$state" || fail "could not acknowledge the intentional phase-A watcher stop" -# The regression the deferral must NOT swallow: after the shared daemon exits, the -# run record it left behind is never advanced, so `axi status` keeps reporting -# running/fixing and fm-crew-state.sh keeps reporting a working run-step - exactly -# the stale-record hazard rule 7 of the generated brief makes crews check for. The -# record alone therefore proves nothing; without the daemon probe answering up, -# the pane must still reach the unchanged escalation ladder. -test_busy_pane_with_dead_daemon_still_escalates_past_turn_age_bound() { - local dir state fakebin out drain_out capture_file window key pane_hash sig pid wt back - dir=$(make_case busy-dead-daemon); state="$dir/state"; fakebin="$dir/fakebin" - out="$dir/watch.out"; drain_out="$dir/drain.out"; capture_file="$dir/pane.txt" - window="test:fm-dead-daemon"; wt="$dir/wt"; mkdir -p "$wt" - printf 'Working... (4210.6s)' > "$capture_file" - printf 'window=%s\nkind=ship\nharness=pi\nworktree=%s\n' "$window" "$wt" > "$state/dead-daemon.meta" - record_pi_busy "$state" dead-daemon - printf 'working: validating\n' > "$state/dead-daemon.status" - sig=$(seen_sig "$state/dead-daemon.status"); printf '%s' "$sig" > "$state/.seen-dead-daemon_status" - key=$(printf '%s' "$window" | tr ':/.' '___') - pane_hash=$(hash_text "Working... (4210.6s)") - printf '%s' "$pane_hash" > "$state/.hash-$key" - printf '1\n' > "$state/.count-$key" - touch -t 200001010000 "$state/dead-daemon.turn-ended" - prime_turnend_seen "$state/dead-daemon.turn-ended" - back=$(( $(date +%s) - 500 )) + # Phase B: same window, same busy pane, same attributed working run-step - but + # the run has gone quiet (a hung step, or a record left behind after the daemon + # exited), so fm-crew-state.sh no longer marks its activity recent. echo "$back" > "$state/.stale-since-$key" set_mtime "$back" "$state/.stale-since-$key" - - # Same still-running record as the deferral above; only the daemon probe differs. + : > "$out" PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ - FM_FAKE_NM_DAEMON=down \ FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ FM_PAUSE_RESURFACE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! - wait_for_exit "$pid" 100 || fail "a stale running record after a dead daemon did not wedge-escalate: $(cat "$out")" - grep -F "stale: $window" "$out" >/dev/null || fail "the dead-daemon escalation did not print the stale wake" - grep -F "possible wedge" "$out" >/dev/null || fail "the dead-daemon escalation did not flag a possible wedge" - [ "$(cat "$state/.wedge-escalations-$key" 2>/dev/null || true)" = 1 ] || fail "the dead-daemon escalation was not counted" - [ ! -e "$state/.nmrun-since-$key" ] || fail "a dead daemon still opened a live-validation deferral chain" - FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || fail "drain after the dead-daemon escalation failed" - grep "$(printf '\tstale\t')" "$drain_out" | grep -F "$window" >/dev/null || fail "the dead-daemon escalation was not queued" - pass "a busy worker whose daemon is down still escalates past the completed-turn bound, however live its run record reads" -} - -# A deferral is not silence here either: a validation can hold a live record while -# making no progress, so the whole live-run deferral chain ages and re-surfaces -# once per PAUSE_RESURFACE_SECS - the same bounded cadence a declared pause and a -# write deferral use - labeled as a recheck rather than a wedge. -test_live_validation_deferral_resurfaces_on_the_bounded_cadence() { - local dir state fakebin out drain_out capture_file window key pane_hash sig pid wt back - dir=$(make_case busy-live-validation-resurface); state="$dir/state"; fakebin="$dir/fakebin" - out="$dir/watch.out"; drain_out="$dir/drain.out"; capture_file="$dir/pane.txt" - window="test:fm-long-validation"; wt="$dir/wt"; mkdir -p "$wt" - printf 'Working... (9000.2s)' > "$capture_file" - printf 'window=%s\nkind=ship\nharness=pi\nworktree=%s\n' "$window" "$wt" > "$state/long-validation.meta" - record_pi_busy "$state" long-validation - printf 'working: validating\n' > "$state/long-validation.status" - sig=$(seen_sig "$state/long-validation.status"); printf '%s' "$sig" > "$state/.seen-long-validation_status" - key=$(printf '%s' "$window" | tr ':/.' '___') - pane_hash=$(hash_text "Working... (9000.2s)") - printf '%s' "$pane_hash" > "$state/.hash-$key" - printf '1\n' > "$state/.count-$key" - touch -t 200001010000 "$state/long-validation.turn-ended" - prime_turnend_seen "$state/long-validation.turn-ended" - back=$(( $(date +%s) - 500 )) - echo "$back" > "$state/.stale-since-$key" - set_mtime "$back" "$state/.stale-since-$key" - # This pane has been deferring on the live run for 500s already. - : > "$state/.nmrun-since-$key" - set_mtime "$back" "$state/.nmrun-since-$key" - - PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ - FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ - FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ - FM_FAKE_NM_DAEMON=up \ - FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ - FM_PAUSE_RESURFACE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ - FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & - pid=$! - wait_for_exit "$pid" 100 || fail "a long live-validation deferral never re-surfaced on the bounded cadence" - grep -F "stale: $window" "$out" >/dev/null || fail "the live-validation recheck did not print a stale wake" - grep -F "no-mistakes validation proven live" "$out" >/dev/null || fail "the live-validation recheck was not labeled as such" - grep -F "possible wedge" "$out" >/dev/null && fail "a live-validation recheck was mislabeled a possible wedge" - [ -e "$state/.nmrun-resurfaced-$key" ] || fail "the live-validation re-surface throttle marker was not recorded" - [ ! -e "$state/.wedge-escalations-$key" ] || fail "a live-validation recheck advanced the wedge escalation counter" - FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || fail "drain after the live-validation recheck failed" - grep "$(printf '\tstale\t')" "$drain_out" | grep -F "$window" >/dev/null || fail "the live-validation recheck was not queued" - pass "a live-validation deferral re-surfaces once on the bounded pause cadence, so a run that stops advancing cannot stay invisible" + wait_for_exit "$pid" 100 || fail "a quiet run behind a busy pane did not wedge-escalate past the turn-age bound: $(cat "$out")" + grep -F "stale: $window" "$out" >/dev/null || fail "the quiet-run escalation did not print the stale wake" + grep -F "possible wedge" "$out" >/dev/null || fail "the quiet-run escalation did not flag a possible wedge" + [ "$(cat "$state/.wedge-escalations-$key" 2>/dev/null || true)" = 1 ] || fail "the quiet-run escalation was not counted" + [ ! -e "$state/.stale-since-$key" ] || fail "the idle timer was not cleared after a real escalation" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || fail "drain after the quiet-run escalation failed" + grep "$(printf '\tstale\t')" "$drain_out" | grep -F "$window" >/dev/null || fail "the quiet-run escalation was not queued" + pass "a busy worker's wedge escalation is deferred only while its own no-mistakes run reports recent activity" } # --- declared pause + busy pane: the busy-turn bound must honor the declaration @@ -4575,9 +4508,7 @@ test_busy_pane_stable_hash_escalates_past_turn_age_bound test_busy_pane_changing_hash_escalates_past_turn_age_bound test_busy_pane_turn_end_touch_resets_age test_busy_pane_repeated_escalation_reaches_demand_deep_inspection -test_busy_pane_with_live_validation_defers_the_wedge_escalation -test_busy_pane_with_dead_daemon_still_escalates_past_turn_age_bound -test_live_validation_deferral_resurfaces_on_the_bounded_cadence +test_busy_turn_bound_defers_only_while_the_run_reports_recent_activity test_busy_pane_default_turn_age_bound_is_3600s test_busy_declared_pause_is_rechecked_not_wedge_escalated test_afk_busy_declared_pause_hands_off_plain_stale diff --git a/tests/wake-helpers.sh b/tests/wake-helpers.sh index d9a988b2d38..da83bb3dc91 100644 --- a/tests/wake-helpers.sh +++ b/tests/wake-helpers.sh @@ -101,7 +101,6 @@ exit 1 SH chmod +x "$fakebin/tmux" make_fake_crew_state "$fakebin" >/dev/null - make_fake_no_mistakes "$fakebin" >/dev/null printf '%s\n' "$dir" } @@ -130,29 +129,6 @@ SH printf '%s\n' "$fakebin/fm-crew-state.sh" } -# Install a hermetic fake `no-mistakes` into and echo its path, so no -# test ever probes the machine's real shared daemon. Only `daemon status` is -# exercised (the liveness probe behind the wedge deferral for a live validation); -# FM_FAKE_NM_DAEMON=up makes the probe prove the daemon up, and anything else - -# including the default - leaves it unproven, the fail-closed answer a test that -# forgets to set one should get. -make_fake_no_mistakes() { # - local fakebin=$1 - cat > "$fakebin/no-mistakes" <<'SH' -#!/usr/bin/env bash -set -u -if [ "${1:-}" = daemon ] && [ "${2:-}" = status ]; then - case "${FM_FAKE_NM_DAEMON:-down}" in - up) echo "daemon: running"; exit 0 ;; - *) echo "daemon: not running" >&2; exit 1 ;; - esac -fi -exit 1 -SH - chmod +x "$fakebin/no-mistakes" - printf '%s\n' "$fakebin/no-mistakes" -} - # Prime 's .seen-* marker to its CURRENT signature through the production # signature owner (bin/fm-wake-lib.sh), so a test can declare "everything in # this file was already surfaced or deliberately absorbed" before exercising From 470fd993e359f6d3773bac7cd3aa8d5f2380e3ad Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 16:58:27 +0200 Subject: [PATCH 08/11] no-mistakes(review): Allow stopping at a genuine ask-user gate --- bin/fm-dod-lib.sh | 2 +- tests/fm-brief.test.sh | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index b3c304b3dda..312b238cc5e 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -243,7 +243,7 @@ A killed or timed-out call is never evidence the daemon died: the daemon accepts Reattach and keep going rather than reporting the pipeline blocked; rule 7 owns the checks that decide when a pipeline block is real. After every \`no-mistakes axi respond\`, continue in the same turn with bounded calls to the structured \`no-mistakes axi status\` interface until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block. An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run. -Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block. +Never end your turn or promise to resume or check later while structured status shows that validation is active, unless the attributed run presents a genuine ask-user decision - escalate it and stop - or rule 7's daemon checks have established a real block. Two firstmate-specific rules layer on top of that guidance: - ask-user findings are never yours to answer: escalate to firstmate using rule 6's ask-user format and stop. diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 9567f7fd292..1c3a8277627 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -388,7 +388,7 @@ test_active_no_mistakes_validation_cannot_be_deferred() { "no-mistakes brief did not define the only status-polling stop conditions" assert_grep "An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run." "$brief" \ "no-mistakes brief did not extend the continuation rule past a gate response to starting and reattaching" - assert_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block." "$brief" \ + assert_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active, unless the attributed run presents a genuine ask-user decision - escalate it and stop - or rule 7's daemon checks have established a real block." "$brief" \ "no-mistakes brief still permits deferring an active validation run" pass "fm-brief.sh: active no-mistakes validation continues in the same turn through the next real transition" } @@ -420,7 +420,7 @@ test_direct_pr_requires_forge_proof_and_diagnosis() { for unaffected in "$scout" "$charter" "$local_brief"; do assert_no_grep "diagnose the forge failure first" "$unaffected" \ "an unaffected scaffold received the direct-PR forge contract" - assert_no_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active and rule 7's daemon checks have not established a real block." "$unaffected" \ + assert_no_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active, unless the attributed run presents a genuine ask-user decision - escalate it and stop - or rule 7's daemon checks have established a real block." "$unaffected" \ "an unaffected scaffold received the no-mistakes active-run contract" done pass "fm-brief.sh: direct-PR completion requires forge diagnosis, a pushed branch, and a verified full URL" From ad3e90d78fd13bff5484fc19dc1197ce192e5132 Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 17:23:16 +0200 Subject: [PATCH 09/11] no-mistakes(document): document busy-turn live-run deferral in architecture and config --- bin/fm-crew-state.sh | 6 ++++++ docs/architecture.md | 5 ++++- docs/configuration.md | 2 +- 3 files changed, 11 insertions(+), 2 deletions(-) diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index fc4924f06a7..cb7f78af38d 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -61,6 +61,12 @@ # FAILED record whose daemon an explicit probe proves down reads unknown, # never failed: an instrument failure must not read as work failure # (nm_daemon_probe_down). +# A working run-step also carries a positive `run activity recent` note in +# its detail while the pipeline's own recency verdict says an active step +# is still reporting (nm_run_activity_is_recent, which requires the +# captured `active_steps[]` table and never treats an absent one as +# recency). Supervisors read that note to tell an advancing run from a +# record nothing is executing. # 3. Reconcile the status log: if its last line says needs-decision/blocked but # the run-step shows the run moved on, the log is deterministically stale and # is flagged superseded. A genuinely parked run plus a needs-decision log diff --git a/docs/architecture.md b/docs/architecture.md index 58b900786d4..29d6640a936 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -22,8 +22,11 @@ That deferral re-surfaces on the same `FM_PAUSE_RESURFACE_SECS` cadence as a dec Every absence of write evidence, including a missing worktree record, a torn-down worktree, a walk that outlives its wall-clock bound on a hung mount, and a failed walk, leaves the existing escalation schedule untouched, so a crew that writes nothing still escalates exactly as before. A secondmate's recorded worktree is never probed for write activity, because it is a provisioned firstmate home whose own supervision keeps writing inside it whether or not the mate produces anything, so its panes keep escalating on the unchanged schedule. A busy pane is otherwise exempt from staleness, but only until its latest `state/.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, worktree-write deferral, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. -A crew that declared an external wait (`paused:`) or a verified captain-held transfer is the one exception to that bound: its busy verdict supplies liveness while identifying the long-running foreground call as the declared wait, so it takes the bounded `FM_PAUSE_RESURFACE_SECS` recheck instead of a wedge escalation. +A crew that declared an external wait (`paused:`) or a verified captain-held transfer is one exception to that bound: its busy verdict supplies liveness while identifying the long-running foreground call as the declared wait, so it takes the bounded `FM_PAUSE_RESURFACE_SECS` recheck instead of a wedge escalation. Lifting the declaration restores the unchanged busy-pane wedge path, while a pane that is no longer busy returns to the existing idle declared-wait classification. +The other exception is a worker inside its own no-mistakes validation, whose contract holds one turn open for the whole run so its completed-turn age is expected to cross the bound: at each `FM_STALE_ESCALATE_SECS` threshold, a busy pane whose own attributed run reports recent activity defers that one escalation and restarts the idle timer, so the next threshold has to prove the activity again. +The proof is the pipeline's own recency verdict on the step attributed to this crew, read through `bin/fm-crew-state.sh` only in the branch that was about to escalate rather than on every poll, and never the liveness of the shared no-mistakes daemon, which any other crew's run keeps up. +A hung step, a record left behind after the daemon exited under it, and a run that is no longer this crew's all stop reporting that activity, so the threshold falls straight through to the unchanged escalation ladder and its `demand-deep-inspection` marker; this deferral exists only on the busy-pane path, and the non-busy stale classification is unchanged. While away mode is active, a busy pane that crosses the bound under a declared wait is handed to the daemon as the plain wake identity instead of taking that recheck in the watcher, because the daemon owns triage there and a wake already decorated as a possible wedge would override the daemon's own declared-wait verdict; an undeclared busy pane past the bound still takes the wedge escalation in away mode. That handoff is keyed on the declaration itself (the status log's signature) rather than on the pane capture, so a harness footer that ticks on every poll wakes the daemon once per declaration instead of once per poll, and it clears the wedge timer, escalation count, and worktree-write deferral exactly as the normal-mode absorber does, so an undeclared busy phase's timer does not resume when the declaration lifts. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) only after generation-bound recovery evidence is published, so an interrupted watcher or handling turn can be recovered without losing the queue record. diff --git a/docs/configuration.md b/docs/configuration.md index ffe847b7f10..78048136faf 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -931,7 +931,7 @@ FM_TURNEND_CHURN_ABSORB_SECS=900 # longest one endpoint's bare turn-ends may b FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless admitted directly to the declared-wait cadence, while a live idle declared wait still surfaces once before that cadence bounds repeats -FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/.turn-ended marker, or its state/.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait or verified captain-held transfer takes the FM_PAUSE_RESURFACE_SECS recheck below instead +FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/.turn-ended marker, or its state/.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait or verified captain-held transfer takes the FM_PAUSE_RESURFACE_SECS recheck below instead, and a busy pane whose own attributed no-mistakes run reports recent activity defers one escalation per FM_STALE_ESCALATE_SECS threshold and must prove that activity again for the next one FM_PAUSE_RESURFACE_SECS=3600 # seconds between bounded rechecks of a declared external wait or verified captain-held transfer, and between repeated new-hash stale alarms for an ordinary crew task with an open backlog captain call; this includes a live idle pane after its first inconclusive stale wake and a live busy pane past FM_BUSY_TURN_MAX_SECS, while the away-mode daemon uses the same setting and ages its window against the crew's own latest status line rather than pane busy state FM_SECONDMATE_WAKE_STALL_SECS=180 # minimum interval with no change of the oldest actionable foreign wake-queue row (it advances as the mate drains, and a queue reprovisioned under the same task id starts a fresh interval at whatever sequence it restarts) before an endpoint-recorded local secondmate produces one durable parent wake-loop-stall notification for that no-progress episode; a mate that is provably inside an active turn (an exact busy verdict, bounded by the same FM_BUSY_TURN_MAX_SECS above) never escalates whatever this interval says, declared external-wait pause rows are excluded, and zero or invalid values use 180 FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added From cd3f4becc260ed6dc7ee9ff2cb65a2e4c9ec17e1 Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 17:48:25 +0200 Subject: [PATCH 10/11] no-mistakes(ci): The failing check "Behavior tests (Herdr)" came from tests/fm-backend-herdr-focus-flash-e2e.test.sh Part C, a file untouched by this PR, but it is a real flaky test rather than an external/infra failure, so I made it deterministic. Root cause: on a below-floor herdr release (CI runs 0.7.4, steal_live=1), Part C required an external polling sampler to catch at least one wrong-focus sample inside the window between the plain explicit close and the adapter's restore backstop. That window is a couple of CLI round-trips wide while each sampler iteration costs two lab calls, so catching a sample is pure timing: CI caught zero and failed ("observed no wrong-focus sample at all, so the sampler proved nothing"); the identical code caught 4 samples locally. Fix: Part C now reads the same fact from deterministic in-band evidence it already collects, the adapter call log. fm_backend_herdr_projection_focus_restore issues `tab focus` exactly when its own post-close snapshot differs from the pre-operation snapshot, so on a defective release `^tab focus` must appear (wrong-focus window existed and the exact-tab backstop closed it, alongside the pre-existing C_AFTER = C_BEFORE outcome assertion), and on a focus-preserving release it must not appear. This mirrors the call-log assertion Part B already makes, so the racy Part C sampler (subshell, ready/active/stop files, sample tallies) was deleted rather than tuned. Part B's sampler and all production code are untouched; no new machinery added. Verified locally with real herdr 0.7.4 (temporarily started a headless default session for the lab tripwire, then `herdr server stop`; default session back to not-running, other sessions untouched): the test passed 3 consecutive runs with exit 0, and bin/fm-lint.sh (ShellCheck 0.11.0 + actionlint) is clean --- .../fm-backend-herdr-focus-flash-e2e.test.sh | 46 ++++--------------- 1 file changed, 10 insertions(+), 36 deletions(-) diff --git a/tests/fm-backend-herdr-focus-flash-e2e.test.sh b/tests/fm-backend-herdr-focus-flash-e2e.test.sh index 69ec69e8224..0a8a7206f4f 100755 --- a/tests/fm-backend-herdr-focus-flash-e2e.test.sh +++ b/tests/fm-backend-herdr-focus-flash-e2e.test.sh @@ -266,33 +266,12 @@ while [ "$C_CHILD_ATTEMPT" -lt 100 ]; do done [ "$C_CHILD_STABLE" -ge 2 ] || fail 'the Part C doomed pane never reported a stable persistent child process' +# The plain close's wrong-focus window is bounded by the operation itself, so +# it is read from the call log rather than raced with an external sampler: the +# corrective `tab focus` is issued exactly when the adapter's own post-close +# snapshot differs from the pre-operation one. C_CALL_LOG="$TMP_ROOT/call-c.log" -C_FOCUS_SAMPLES="$TMP_ROOT/focus-c.samples" -C_OPERATION_ACTIVE="$TMP_ROOT/operation-c.active" -C_SAMPLER_READY="$TMP_ROOT/sampler-c.ready" -SAMPLER_STOP="$TMP_ROOT/sampler-c.stop" : > "$C_CALL_LOG" -: > "$C_FOCUS_SAMPLES" -( - : > "$C_SAMPLER_READY" - while [ ! -e "$SAMPLER_STOP" ]; do - if [ -e "$C_OPERATION_ACTIVE" ]; then - if C_SAMPLE=$(focus_snapshot); then - printf '%s\n' "$C_SAMPLE" >> "$C_FOCUS_SAMPLES" - else - printf '%s\n' UNREADABLE >> "$C_FOCUS_SAMPLES" - fi - fi - done -) & -SAMPLER_PID=$! -C_READY_ATTEMPT=0 -while [ ! -e "$C_SAMPLER_READY" ] && [ "$C_READY_ATTEMPT" -lt 100 ]; do - sleep 0.01 - C_READY_ATTEMPT=$((C_READY_ATTEMPT + 1)) -done -[ -e "$C_SAMPLER_READY" ] || fail 'the Part C focus sampler did not start' -: > "$C_OPERATION_ACTIVE" # A short proof budget keeps the exhausted-proof path fast; the count below is # what proves the proof was exhausted rather than skipped. C_PROOF_POLLS=3 @@ -308,10 +287,6 @@ C_OUT=$(PATH="$FAKEBIN:$HERDR_ORIGINAL_PATH" FM_FLASH_CALL_LOG="$C_CALL_LOG" \ fm_backend_herdr_projection_close_pane_focus_preserving "$2" "$3" ' _ "$ROOT" "$HERDR_LAB_SESSION" "$C_DOOMED_PANE" 2>&1) C_STATUS=$? -rm -f "$C_OPERATION_ACTIVE" -: > "$SAMPLER_STOP" -wait "$SAMPLER_PID" 2>/dev/null || true -SAMPLER_PID= [ "$C_STATUS" -eq 0 ] || fail "the production focus-preserving close failed (status $C_STATUS): $C_OUT" wait_ws_gone "$C_DOOMED_WS" || fail 'the fallback close left the doomed workspace behind' if lab pane get "$C_DOOMED_PANE" >/dev/null 2>&1; then @@ -333,19 +308,18 @@ pass 'fallback: a doomed pane holding a persistent child exhausts the proof and C_AFTER=$(focus_snapshot) || fail 'could not capture the Part C post-close focus' [ "$C_AFTER" = "$C_BEFORE" ] \ || fail "the fallback close left focus off the anchor ($C_BEFORE -> $C_AFTER)" -C_WRONG=$(grep -Fvxc -- "$C_BEFORE" "$C_FOCUS_SAMPLES" || true) if [ "$STEAL_LIVE" = 1 ]; then # A defective release cannot make this path focus-safe, which is precisely why # default-on projection is floored above it. The wrong-focus window is # explicitly accepted here, but only as a BOUNDED one: the restore backstop - # must have put the anchor back exactly, and the whole exposure must end with + # must have fired and put the anchor back exactly, so the exposure ends with # the operation rather than parking the captain somewhere else. - [ "$C_WRONG" -ge 1 ] \ - || fail 'Part C reached the fallback on a defective release but observed no wrong-focus sample at all, so the sampler proved nothing' - pass "fallback on a defective release: a bounded wrong-focus window of $C_WRONG samples was fully restored to the anchor" + grep -q '^tab focus' "$C_CALL_LOG" \ + || fail 'Part C took the fallback on a defective release without the corrective restore, so no bounded wrong-focus window was exercised' + pass 'fallback on a defective release: the wrong-focus window the plain close opens was closed by the exact-tab restore' else - [ "$C_WRONG" -eq 0 ] \ - || fail "a focus-preserving release exposed $C_WRONG wrong-focus samples on the fallback path" + grep -q '^tab focus' "$C_CALL_LOG" \ + && fail 'a focus-preserving release needed the corrective restore, so the fallback path exposed a wrong-focus window' pass 'fallback on a focus-preserving release: the plain explicit close preserved exact focus throughout' fi From 9707d7698b7e9edec23c78e778d20ecf512dd6af Mon Sep 17 00:00:00 2001 From: Alex William Date: Tue, 8 Sep 2026 18:44:17 +0200 Subject: [PATCH 11/11] no-mistakes(document): Clarify watcher deferrals and focus verification evidence --- docs/architecture.md | 2 +- docs/configuration.md | 2 +- docs/verification/runtime-backends.md | 5 ++++- 3 files changed, 6 insertions(+), 3 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index 29d6640a936..eeefcd2d81f 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -27,7 +27,7 @@ Lifting the declaration restores the unchanged busy-pane wedge path, while a pan The other exception is a worker inside its own no-mistakes validation, whose contract holds one turn open for the whole run so its completed-turn age is expected to cross the bound: at each `FM_STALE_ESCALATE_SECS` threshold, a busy pane whose own attributed run reports recent activity defers that one escalation and restarts the idle timer, so the next threshold has to prove the activity again. The proof is the pipeline's own recency verdict on the step attributed to this crew, read through `bin/fm-crew-state.sh` only in the branch that was about to escalate rather than on every poll, and never the liveness of the shared no-mistakes daemon, which any other crew's run keeps up. A hung step, a record left behind after the daemon exited under it, and a run that is no longer this crew's all stop reporting that activity, so the threshold falls straight through to the unchanged escalation ladder and its `demand-deep-inspection` marker; this deferral exists only on the busy-pane path, and the non-busy stale classification is unchanged. -While away mode is active, a busy pane that crosses the bound under a declared wait is handed to the daemon as the plain wake identity instead of taking that recheck in the watcher, because the daemon owns triage there and a wake already decorated as a possible wedge would override the daemon's own declared-wait verdict; an undeclared busy pane past the bound still takes the wedge escalation in away mode. +While away mode is active, a busy pane that crosses the bound under a declared wait is handed to the daemon as the plain wake identity instead of taking that recheck in the watcher, because the daemon owns triage there and a wake already decorated as a possible wedge would override the daemon's own declared-wait verdict; an undeclared busy pane past the bound still takes the wedge path in away mode, including the evidence-based deferrals above. That handoff is keyed on the declaration itself (the status log's signature) rather than on the pane capture, so a harness footer that ticks on every poll wakes the daemon once per declaration instead of once per poll, and it clears the wedge timer, escalation count, and worktree-write deferral exactly as the normal-mode absorber does, so an undeclared busy phase's timer does not resume when the declaration lifts. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) only after generation-bound recovery evidence is published, so an interrupted watcher or handling turn can be recovered without losing the queue record. Agent endpoint liveness and queue-consumption liveness are separate: on each poll, the primary watcher reads the oldest valid actionable row from every endpoint-recorded local secondmate home's durable wake queue without locking, consuming, or rewriting that foreign queue. diff --git a/docs/configuration.md b/docs/configuration.md index 78048136faf..868daa9fee8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -931,7 +931,7 @@ FM_TURNEND_CHURN_ABSORB_SECS=900 # longest one endpoint's bare turn-ends may b FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless admitted directly to the declared-wait cadence, while a live idle declared wait still surfaces once before that cadence bounds repeats -FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/.turn-ended marker, or its state/.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait or verified captain-held transfer takes the FM_PAUSE_RESURFACE_SECS recheck below instead, and a busy pane whose own attributed no-mistakes run reports recent activity defers one escalation per FM_STALE_ESCALATE_SECS threshold and must prove that activity again for the next one +FM_BUSY_TURN_MAX_SECS=3600 # busy-turn age bound in seconds; docs/architecture.md "Event-driven supervision" owns the age sources, inspection-only escalation, and declared-wait and recent-run-activity deferrals FM_PAUSE_RESURFACE_SECS=3600 # seconds between bounded rechecks of a declared external wait or verified captain-held transfer, and between repeated new-hash stale alarms for an ordinary crew task with an open backlog captain call; this includes a live idle pane after its first inconclusive stale wake and a live busy pane past FM_BUSY_TURN_MAX_SECS, while the away-mode daemon uses the same setting and ages its window against the crew's own latest status line rather than pane busy state FM_SECONDMATE_WAKE_STALL_SECS=180 # minimum interval with no change of the oldest actionable foreign wake-queue row (it advances as the mate drains, and a queue reprovisioned under the same task id starts a fresh interval at whatever sequence it restarts) before an endpoint-recorded local secondmate produces one durable parent wake-loop-stall notification for that no-progress episode; a mate that is provably inside an active turn (an exact busy verdict, bounded by the same FM_BUSY_TURN_MAX_SECS above) never escalates whatever this interval says, declared external-wait pause rows are excluded, and zero or invalid values use 180 FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index f8e63fa8ca4..1cbf3c4575c 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -818,7 +818,10 @@ The same guarded named-lab command passed on 2026-09-03 against Herdr 0.8.2 afte It reported `steal_live=0 floor_verdict=0 default-session-tripwire=armed`, with the fleet's default session unchanged before and after. Part C is the case the suite could not reach before: a doomed pane whose shell holds a persistent background child fails the lone-idle-shell proof on every sample, so the plan takes the plain explicit close, in the geometry where the closing workspace's right neighbour is a spacer rather than the focused anchor. -On 0.7.5 that fallback exposed a bounded four-sample wrong-focus window and restored the anchor exactly; on 0.8.0 the same fallback exposed none, which is why default-on projection is floored at 0.8.0 rather than mitigated further below it. +In the recorded runs above, the sampler observed a bounded four-sample wrong-focus window on 0.7.5 with exact anchor restoration and none on 0.8.0, supporting the 0.8.0 default-on floor. +The current Part C regression instead checks the adapter's call log for corrective `tab focus` and verifies the final exact anchor: correction must occur on a release Part A proves defective and must be absent on a focus-preserving release. +It no longer samples focus concurrently during the fallback close; the recorded sample counts are prior evidence, not output of the current guard. +Part B retains concurrent sampling of the mitigated removal path. The suite also cross-checks its own Part A measurement against the floor classifier on whatever release it runs, so a drifted protocol-to-release mapping fails there rather than silently gating on the wrong thing. ### Presentation version floor