From 294b7f181c3d447102f406d5d39740ab23f5f93e Mon Sep 17 00:00:00 2001 From: cm-maple7 Date: Mon, 31 Aug 2026 20:25:59 -0700 Subject: [PATCH 1/6] fix(bin): stop the watcher wedge-escalating legitimate background waits A no-mistakes background validation run can legitimately sit on an idle pane far longer than STALE_ESCALATE_SECS. wedge_timer_check used to escalate purely on elapsed time once a chain was classified working, so a long validation run got wedge-escalated every few minutes despite the pipeline actively running. It now re-verifies the run-step is still working at each threshold crossing (reverify_working) before escalating, mirroring the existing worktree-write positive-evidence check. Separately, a worker waiting on a captain-reserved PR merge (or on upstream checks that may never post without maintainer approval) declares paused:, but the ci step's own run-step reading otherwise stayed working for the whole wait and outranked that declaration every poll, so a legitimately parked task got wedge-escalated forever. The crew-state reader now reconciles a declared pause against a run whose only remaining step is CI monitoring, reporting it as paused (never overriding a genuine gate or fixing step); the watcher trusts that run-step-attributed paused verdict outright instead of requiring a dead-looking agent pane, while a status-log-only pause (no run to corroborate it) still goes through the existing agent-liveness recovery so a live interactive gate is not silently hidden behind a stale declaration. --- bin/fm-classify-lib.sh | 21 +++++++ bin/fm-crew-state.sh | 24 ++++++- bin/fm-watch.sh | 59 ++++++++++++----- tests/fm-crew-state.test.sh | 41 ++++++++++++ tests/fm-watch-triage.test.sh | 115 ++++++++++++++++++++++++++++++---- 5 files changed, 231 insertions(+), 29 deletions(-) diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 646c399e198..91bdd5a139c 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -1309,6 +1309,27 @@ crew_is_paused() { # [ "$(crew_absorb_class "$1")" = paused ] } +# 0 if crew 's declared pause/captain-held wait is attributed to the +# run-step (fm-crew-state.sh's ci-monitoring-only reconciliation), as opposed +# to the no-run status-log fallback. A run-step pause has already been +# corroborated against the pipeline's own state - no gate, not fixing, no +# independently-resolved outcome - so fm-watch.sh's pause_state_class trusts +# it outright, live agent or not. A no-run fallback pause carries no such +# corroboration and instead still needs pause_state_class's agent-liveness +# recovery: a still-alive agent may be sitting at an undeclared live +# interactive gate rather than the wait it named. A second $FM_CREW_STATE_BIN +# read (crew_absorb_class already made one), but only inside the already-rare +# declared-pause branch its caller gates this behind, never per poll. +crew_run_step_paused() { # + local id=$1 line + [ -n "$id" ] || return 1 + line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || return 1 + case "$line" in + "state: paused"*"source: run-step"*) return 0 ;; + *) return 1 ;; + esac +} + # Directories excluded from the worktree write probe below, and the depth it walks. # The excluded set is everything a supervisor read or a package manager can write # without the crew doing any work - .git first, so firstmate's own read-only git diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index 3566b073191..054c0518552 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -48,7 +48,11 @@ # 3. Reconcile the status log: if its last line says needs-decision/blocked but # the run-step shows the run moved on, the log is deterministically stale and # is flagged superseded. A genuinely parked run plus a needs-decision log -# agree, and are reported as parked. +# agree, and are reported as parked. A last line that instead declares +# paused:/captain-held while the run's only remaining step is CI +# monitoring (never fixing, never a gate) wins over the run-step's own +# working/done reading and is reported as paused, since that is a +# declared external wait, not active work. # 4. No run for this crew (pre-validation, or kind=scout): fall back to the # recorded backend's pane busy state, then the status log's last line only # when its verb maps to a recognized run-state. Decision-only events such as @@ -585,6 +589,24 @@ if [ "$HAVE_RUN" = 1 ]; then ;; esac fi + # A crew whose only remaining pipeline activity is CI monitoring (no + # gate, not fixing, no independently-resolved outcome) is, once it has + # declared paused:/captain-held itself, intentionally waiting on an + # external dependency - upstream checks that may never post without + # maintainer approval, or the captain's own merge - not "still + # validating". Root cause of the 2026-08-27 fm-parked-run-wedge-alarm + # incident: this run-step reading otherwise stays working/done and + # outranks the declared wait every poll, so the watcher wedge-escalates + # a legitimately parked task forever. Restricting the override to the ci + # step specifically (never fixing, never a gate) keeps a genuinely + # active run's own state authoritative - a declared pause there is + # stale and must not silence it. + if [ "$CI_STEP_STATUS" = running ] \ + && [ "$has_gate" = 0 ] && [ -z "$outcome" ] \ + && status_is_paused_or_captain_held "$LOG_LINE"; then + RUN_STATE=paused + RUN_DETAIL="declared wait (ci monitoring): $RUN_DETAIL" + fi fi fi diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index a8cf1610686..0b557b9c395 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -538,15 +538,25 @@ clear_write_tracking() { # # absorbed as provably-working - repairs a missing/corrupt timer (self-heals a # watcher restart between recording the hash and recording the timer), or # escalates once STALE_ESCALATE_SECS have elapsed. Never re-reads the crew -# state (the costly check already ran once, at classification time). Shared by -# both places a hash can be absorbed this way: the plain non-terminal path, -# and the stale_is_terminal-overridden path (a captain-relevant status-log -# line that an active run/busy pane outranked). -# The worktree write probe runs ONLY here, inside the at-threshold branch that is -# about to escalate: at most one bounded walk per window per STALE_ESCALATE_SECS, -# never per poll. -wedge_timer_check() { # - local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 since age n reason +# state by default (the costly check already ran once, at classification +# time), EXCEPT when the caller passes reverify_working=1 (the chain was +# opened by a run-step "working" classification rather than a busy pane): a +# no-mistakes background validation can legitimately sit on an idle pane far +# longer than STALE_ESCALATE_SECS, so at each threshold crossing this +# re-confirms the run is still provably working before escalating - the same +# "never escalate past positive evidence" contract crew_worktree_written_since +# already gives a busy foreground pane, applied to a background run's own +# authoritative run-step reading instead. A run-step that has moved to parked +# (awaiting_approval/fix_review) or otherwise stopped reporting working no +# longer re-verifies true, so it still escalates rather than absorbing forever. +# Shared by both places a hash can be absorbed this way: the plain +# non-terminal path, and the stale_is_terminal-overridden path (a +# captain-relevant status-log line that an active run/busy pane outranked). +# The worktree write probe and the reverify check both run ONLY here, inside +# the at-threshold branch that is about to escalate: at most one bounded check +# per window per STALE_ESCALATE_SECS, never per poll. +wedge_timer_check() { # [reverify_working] + local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 reverify_working=${6:-} since age n reason since=$(cat "$since_file" 2>/dev/null || true) case "$since" in ''|*[!0-9]*) @@ -559,6 +569,11 @@ wedge_timer_check() { # "$since_file" + triage_log "absorbed $label timer reset (re-verified still working): $win" + return 0 + fi if crew_worktree_written_since "$task" "$STATE" "$since_file"; then wedge_defer_writing "$win" "$since_file" "$label" "$age" return 0 @@ -699,9 +714,18 @@ clear_pause_tracking() { # } # Reconcile a declared pause or captain-held status with authoritative crew state. -# After fm-crew-state has fallen back to stopped or unknown, paused classification is -# recovered only for a confidently dead ordinary crew, or for a secondmate, whose -# endpoint liveness this function deliberately never reads. +# A run-step-attributed paused verdict (fm-crew-state.sh's ci-monitoring-only +# reconciliation, which has already corroborated the wait against the +# pipeline's own state: no gate, not fixing, no independently-resolved +# outcome) is trusted immediately, live agent or not - a worker's own +# supervising loop staying alive to poll CI in the background is expected, +# not evidence the declared wait is stale. After fm-crew-state has fallen +# back to stopped or unknown, or the pause came only from the status log with +# no run to corroborate it, paused classification instead requires a +# confidently dead ordinary crew's agent (or a secondmate, whose endpoint +# liveness this function deliberately never reads): a status-log pause from a +# still-alive agent may be sitting at an undeclared live interactive gate +# rather than the wait it named, so it is surfaced instead of trusted. pause_state_class() { # local win=$1 task=$2 key last recheck_file class agent_alive kind key=$(window_key "$win") @@ -712,6 +736,11 @@ pause_state_class() { # crew_absorb_class "$task" return fi + if crew_run_step_paused "$task"; then + date +%s > "$recheck_file" + printf 'paused' + return + fi # Read once past the declared-wait gate and reused by both liveness gates below, # so a mate's stale poll costs one metadata scan rather than one per gate, and the # far more common no-declaration path above still costs none. @@ -1457,7 +1486,7 @@ EOF # wedge timer is running for it) - keep treating it that way # without re-reading the crew state every poll, and without # letting the still-captain-relevant log line re-surface it. - wedge_timer_check "$w" "$ssf" "stale (overridden terminal status)" "$ewf" "$task" + wedge_timer_check "$w" "$ssf" "stale (overridden terminal status)" "$ewf" "$task" 1 fi # else: already surfaced as genuinely terminal on a prior poll of # this same hash - nothing left to do (matches the original, @@ -1500,12 +1529,12 @@ EOF paused) handle_paused_stale "$w" "$task" "$h" ;; working) clear_pause_state "$key" printf '%s' "$h" > "$sf" - wedge_timer_check "$w" "$ssf" "non-terminal stale (provably working after a declared pause)" "$ewf" "$task" + wedge_timer_check "$w" "$ssf" "non-terminal stale (provably working after a declared pause)" "$ewf" "$task" 1 triage_log "absorbed non-terminal stale (provably working): $w" ;; *) handle_paused_stale "$w" "$task" "$h" ;; esac else - wedge_timer_check "$w" "$ssf" "non-terminal stale" "$ewf" "$task" + wedge_timer_check "$w" "$ssf" "non-terminal stale" "$ewf" "$task" 1 fi fi fi diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 16f991b5c0b..9821a632ca3 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -576,6 +576,45 @@ test_ci_monitoring_still_waiting_stays_working() { pass "ci-monitoring run with checks not yet green stays working" } +# Regression for the 2026-08-27 fm-parked-run-wedge-alarm incident: a worker +# waiting on the captain's own merge (or on upstream checks that may never +# post without maintainer approval) declares paused:, but the ci step's own +# run-step reading otherwise stays "working" for the entire wait and used to +# outrank the declaration every poll. The declared wait must win once the +# run's only remaining activity is CI monitoring. +test_ci_monitoring_declared_pause_wins_over_working() { + reset_fakes + local d; d=$(new_case ci-waiting-paused) + make_repo_on_branch "$d/wt" fm/feat-ciwaitpaused + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-ciwaitpaused.meta" "window=fm:fm-feat-ciwaitpaused" "worktree=$d/wt" "kind=ship" + printf 'paused: PR reserved for the captain to merge\n' > "$d/state/feat-ciwaitpaused.status" + FM_FAKE_AXI_STATUS="$(run_ci_monitoring fm/feat-ciwaitpaused)" + FM_FAKE_CI_LOGS="CI checks running, waiting for results..." + local out; out=$(run_crew_state "$d" feat-ciwaitpaused) + assert_contains "$out" "state: paused" "declared wait during ci monitoring -> paused" + assert_contains "$out" "source: run-step" "declared wait during ci monitoring -> run-step source" + assert_not_contains "$out" "state: working" "declared wait must not stay working" + pass "a declared pause during ci-only monitoring wins over the working run-step reading" +} + +# The same declared pause must NOT win once the run has genuinely active work +# left - a gate awaiting review, in this case - since the declaration there is +# stale, not a legitimate description of the current wait. +test_gated_run_declared_pause_does_not_override() { + reset_fakes + local d; d=$(new_case parked-declared-pause) + make_repo_on_branch "$d/wt" fm/feat-gatepaused + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-gatepaused.meta" "window=fm:fm-feat-gatepaused" "worktree=$d/wt" "kind=ship" + printf 'paused: awaiting the upstream release\n' > "$d/state/feat-gatepaused.status" + FM_FAKE_AXI_STATUS="$(run_parked fm/feat-gatepaused)" + local out; out=$(run_crew_state "$d" feat-gatepaused) + assert_contains "$out" "state: parked" "a stale declared pause must not hide a genuine gate" + assert_not_contains "$out" "state: paused" "a genuine gate must not be reported as a declared wait" + pass "a declared pause does not override a run genuinely parked at a gate" +} + # A later merge-conflict auto-fix round after an earlier green reading must # not be masked: the MOST RECENT marker in the log tail wins. test_ci_monitoring_green_then_new_issue_stays_working() { @@ -1726,6 +1765,8 @@ test_ci_monitoring_no_checks_terminal_surfaces_done test_ci_monitoring_green_then_rearm_stays_working test_ci_monitoring_no_checks_yet_stays_working test_ci_monitoring_still_waiting_stays_working +test_ci_monitoring_declared_pause_wins_over_working +test_gated_run_declared_pause_does_not_override test_ci_monitoring_green_then_new_issue_stays_working test_ci_ready_done_log_relapse_stays_working test_ci_fixing_after_green_stays_working diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index 276857fad12..f3a81c27b1a 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -779,6 +779,14 @@ test_terminal_stale_surfaced() { # immediately surfacing a crew that is actively validating. crew_is_provably_working # must get a chance to override a captain-relevant-but-stale status line, exactly # as it already does for a plain non-terminal one. +# +# Regression for the 2026-08-31 rd-tree-goal-replace-anchor incidents: a +# background no-mistakes validation can legitimately sit on an idle pane far +# longer than STALE_ESCALATE_SECS. wedge_timer_check's reverify_working path +# re-confirms the run-step still reads working at each threshold crossing +# instead of blindly time-escalating, so this stays absorbed for as long as +# the run genuinely is; only once the run-step stops reporting working +# (Phase C) does the ladder still escalate. test_stale_terminal_status_overridden_by_active_run() { local dir state fakebin out drain_out capture_file window key pane_hash sig pid dir=$(make_case terminal-stale-overridden); state="$dir/state"; fakebin="$dir/fakebin" @@ -814,25 +822,51 @@ test_stale_terminal_status_overridden_by_active_run() { reap "$pid" ack_stopped_cycle "$state" || fail "could not acknowledge the intentional phase-A watcher stop" - # Phase B: backdate the idle timer past the threshold; the run genuinely - # wedges and the next poll escalates exactly like the non-terminal case. + # Phase B: backdate the idle timer past the threshold while the run-step + # still reads working - the reverify check re-confirms it and resets the + # timer instead of escalating, no matter how long the background run has + # already sat idle. + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "watcher exited past the threshold while still provably working (should re-verify and absorb): $(cat "$out")" + fi + [ ! -s "$out" ] || fail "a re-verified still-working stale printed a wake reason instead of absorbing" + [ ! -s "$state/.wake-queue" ] || fail "a re-verified still-working stale enqueued a wake instead of absorbing" + [ ! -s "$state/.wedge-escalations-$key" ] || fail "a re-verified still-working stale incremented the escalation count" + [ -s "$state/.stale-since-$key" ] || fail "the timer was cleared instead of reset on re-verified absorb" + [ "$(cat "$state/.stale-since-$key")" -gt "$(( $(date +%s) - 500 ))" ] || fail "the timer was not actually reset on re-verified absorb" + reap "$pid" + ack_stopped_cycle "$state" || fail "could not acknowledge the intentional phase-B watcher stop" + + # Phase C: the run-step stops reporting working (it parked, needing a + # captain decision) - backdating the timer again now escalates, exactly as + # a genuinely wedged run always did. + export FM_FAKE_CREW_STATE='state: parked · source: run-step · parked at gate: 1 finding(s)' echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" : > "$out" PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! - wait_for_exit "$pid" 100 || fail "watcher did not escalate an overridden stale terminal status past the threshold" + wait_for_exit "$pid" 100 || fail "watcher did not escalate once the run-step stopped reporting working" grep -F "stale: $window" "$out" >/dev/null || fail "escalation did not print a stale wake" grep -F "possible wedge" "$out" >/dev/null || fail "escalation did not flag a possible wedge" unset FM_FAKE_CREW_STATE - pass "a stale terminal-looking status is overridden and absorbed while a run is actively working, then wedge-escalated" + pass "a stale terminal-looking status is overridden and absorbed while a run is actively working, re-verified past every threshold instead of blindly wedge-escalating, then escalated once the run-step stops reporting working" } -# --- non-terminal stale, crew provably working: absorbed, then wedge-escalated --- +# --- non-terminal stale, crew provably working: absorbed, then re-verified -- +# indefinitely, then escalated once it actually stops working ---------------- # A provably-working crew (an actively-running pipeline) legitimately sits on a -# static pane (e.g. waiting on CI), so a non-terminal stale is absorbed and only -# the wedge timer eventually escalates it - the low-churn behavior preserved. +# static pane (e.g. waiting on CI) far longer than STALE_ESCALATE_SECS, so a +# non-terminal stale is absorbed and re-verified at every threshold crossing +# rather than blindly wedge-escalated on elapsed time alone; only once the +# run-step stops reporting working does the wedge timer escalate. test_nonterminal_stale_provably_working_absorbed_then_escalated() { local dir state fakebin out drain_out capture_file window key pane_hash sig pid @@ -867,21 +901,44 @@ test_nonterminal_stale_provably_working_absorbed_then_escalated() { reap "$pid" ack_stopped_cycle "$state" || fail "could not acknowledge the intentional phase-A watcher stop" - # Phase B: backdate the idle timer past the threshold; the next run escalates. - # (The subsequent-sight timer path does not re-read the crew state.) + # Phase B: backdate the idle timer past the threshold while the run-step + # still reads working; the reverify check re-confirms it and resets the + # timer instead of escalating. + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "watcher exited past the threshold while still provably working (should re-verify and absorb): $(cat "$out")" + fi + [ ! -s "$out" ] || fail "a re-verified still-working stale printed a wake reason instead of absorbing" + [ ! -s "$state/.wake-queue" ] || fail "a re-verified still-working stale enqueued a wake instead of absorbing" + [ ! -s "$state/.wedge-escalations-$key" ] || fail "a re-verified still-working stale incremented the escalation count" + [ -s "$state/.stale-since-$key" ] || fail "the timer was cleared instead of reset on re-verified absorb" + [ "$(cat "$state/.stale-since-$key")" -gt "$(( $(date +%s) - 500 ))" ] || fail "the timer was not actually reset on re-verified absorb" + reap "$pid" + ack_stopped_cycle "$state" || fail "could not acknowledge the intentional phase-B watcher stop" + + # Phase C: the run-step stops reporting working (checks failed and the + # pipeline is fixing them) - a genuinely different signature from a quiet + # background wait, so backdating the timer again now escalates. + export FM_FAKE_CREW_STATE='state: unknown · source: none · no current-state source available' echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" : > "$out" PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! - wait_for_exit "$pid" 100 || fail "watcher did not escalate a provably-working non-terminal stale past the threshold" + wait_for_exit "$pid" 100 || fail "watcher did not escalate once the run-step stopped reporting working" grep -F "stale: $window" "$out" >/dev/null || fail "escalation did not print a stale wake" grep -F "possible wedge" "$out" >/dev/null || fail "escalation did not flag a possible wedge" [ ! -e "$state/.stale-since-$key" ] || fail "stale-since timer was not cleared after escalation" FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || fail "drain after the wedge escalation failed" grep "$(printf '\tstale\t')" "$drain_out" | grep -F "$window" >/dev/null || fail "wedge escalation was not queued" - pass "provably-working non-terminal stale is absorbed on first sight, then wedge-escalated past the threshold" + unset FM_FAKE_CREW_STATE + pass "provably-working non-terminal stale is absorbed on first sight, re-verified past every threshold instead of blindly wedge-escalating, then escalated once the run-step stops reporting working" } # --- non-terminal stale, crew NOT provably working: surfaced immediately ------ @@ -1348,17 +1405,44 @@ test_paused_authoritative_working_preserves_wedge_timer() { reap "$pid" ack_stopped_cycle "$state" || fail "could not acknowledge the intentional authoritative-working stop" + # Still genuinely working past the threshold: reverify_working re-confirms + # it and resets the timer instead of escalating, exactly as a plain + # provably-working non-terminal stale now does. + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "authoritative working state escalated instead of re-verifying past the threshold: $(cat "$out")" + fi + [ ! -s "$out" ] || fail "a re-verified authoritative-working stale printed a wake reason instead of absorbing" + [ -s "$state/.stale-since-$key" ] || fail "the timer was cleared instead of reset on re-verified absorb" + [ "$(cat "$state/.stale-since-$key")" -gt "$(( $(date +%s) - 500 ))" ] || fail "the timer was not actually reset on re-verified absorb" + reap "$pid" + ack_stopped_cycle "$state" || fail "could not acknowledge the intentional re-verify watcher stop" + + # The declared pause clears and the run-step stops reporting working: the + # ladder still escalates once there is nothing left to re-verify. (The log + # must also stop declaring paused: here - as long as it still does, + # pause_state_class's declared-wait branch keeps routing an inconclusive + # crew-state read to the bounded pause recheck rather than the ladder, + # exactly as it does for a live external-decision gate.) + printf 'working: upstream landed, resuming\n' > "$state/paused-working.status" + sig=$(seen_sig "$state/paused-working.status"); printf '%s' "$sig" > "$state/.seen-paused-working_status" + export FM_FAKE_CREW_STATE='state: unknown · source: none · no current-state source available' echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" : > "$out" PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & pid=$! - wait_for_exit "$pid" 100 || fail "authoritative working state did not wedge-escalate past the threshold" + wait_for_exit "$pid" 100 || fail "authoritative working state did not wedge-escalate once it stopped reporting working" grep -F "possible wedge" "$out" >/dev/null || fail "authoritative working wedge escalation omitted its reason" [ ! -e "$state/.stale-since-$key" ] || fail "wedge timer remained after authoritative working escalation" unset FM_FAKE_CREW_STATE - pass "a paused status overridden by authoritative working preserves its wedge timer and escalates" + pass "a paused status overridden by authoritative working preserves its wedge timer, keeps re-verifying instead of escalating while genuinely working, then escalates once it stops" } # --- consecutive wedge escalations on the same pane demand deep inspection ---- @@ -1401,6 +1485,11 @@ test_wedge_escalation_marks_demand_deep_inspection_after_threshold() { reap "$pid" ack_stopped_cycle "$state" || fail "could not acknowledge the intentional wedge priming stop" + # The run-step stops reporting working (the run itself has actually stopped, + # not merely gone quiet) - reverify_working's re-check now finds nothing to + # absorb on, so each round genuinely escalates rather than resetting. + export FM_FAKE_CREW_STATE='state: unknown · source: none · no current-state source available' + n=1 while [ "$n" -le 3 ]; do # Backdate the wedge timer past the threshold before each round, mirroring From 181d678dc886f64d30275ccf68e58a1fda58a0bc Mon Sep 17 00:00:00 2001 From: cm-maple7 Date: Mon, 31 Aug 2026 20:48:43 -0700 Subject: [PATCH 2/6] no-mistakes(review): Bound run-step pause check to recheck cadence; add live-agent trust test --- bin/fm-watch.sh | 23 +++++++++++++-------- tests/fm-watch-triage.test.sh | 38 +++++++++++++++++++++++++++++++++++ 2 files changed, 53 insertions(+), 8 deletions(-) diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 0b557b9c395..2d69fdaf17a 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -703,7 +703,7 @@ busy_turn_bound_check() { # local key=$1 - rm -f "$STATE/.paused-$key" "$STATE/.paused-rechecked-$key" "$STATE/.paused-resurfaced-$key" + rm -f "$STATE/.paused-$key" "$STATE/.paused-rechecked-$key" "$STATE/.paused-resurfaced-$key" "$STATE/.paused-run-step-$key" } clear_pause_tracking() { # @@ -727,25 +727,25 @@ clear_pause_tracking() { # # still-alive agent may be sitting at an undeclared live interactive gate # rather than the wait it named, so it is surfaced instead of trusted. pause_state_class() { # - local win=$1 task=$2 key last recheck_file class agent_alive kind + local win=$1 task=$2 key last recheck_file run_step_file class agent_alive kind key=$(window_key "$win") last=$(last_status_line "$STATE/$task.status") recheck_file="$STATE/.paused-rechecked-$key" + run_step_file="$STATE/.paused-run-step-$key" if ! status_is_paused_or_captain_held "$last"; then rm -f "$recheck_file" crew_absorb_class "$task" return fi - if crew_run_step_paused "$task"; then - date +%s > "$recheck_file" - printf 'paused' - return - fi # Read once past the declared-wait gate and reused by both liveness gates below, # so a mate's stale poll costs one metadata scan rather than one per gate, and the # far more common no-declaration path above still costs none. kind=$(window_kind "$win") if [ -e "$STATE/.paused-$key" ] && [ "$(age_of "$recheck_file")" -lt "$STALE_ESCALATE_SECS" ]; then + if [ -e "$run_step_file" ]; then + printf 'paused' + return + fi if [ "$kind" != secondmate ]; then agent_alive=$(fm_backend_agent_alive "$(window_backend "$win")" "$win" 2>/dev/null) || agent_alive=unknown if [ "$agent_alive" != dead ]; then @@ -759,10 +759,17 @@ pause_state_class() { # fi class=$(crew_absorb_class "$task") if [ "$class" = working ]; then - rm -f "$recheck_file" + rm -f "$recheck_file" "$run_step_file" printf 'working' return fi + if [ "$class" = paused ] && crew_run_step_paused "$task"; then + date +%s > "$recheck_file" + : > "$run_step_file" + printf 'paused' + return + fi + rm -f "$run_step_file" if [ "$kind" != secondmate ]; then agent_alive=$(fm_backend_agent_alive "$(window_backend "$win")" "$win" 2>/dev/null) || agent_alive=unknown if [ "$agent_alive" != dead ]; then diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index f3a81c27b1a..8140849120d 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -1176,6 +1176,43 @@ test_exited_declared_pause_is_bounded_but_live_gate_surfaces() { pass "exited declared-pause and captain-held panes use bounded pause cadence while a live decision gate still surfaces once" } +# Contrasts directly with the alive-decision-gate case above: there, a status-log-only +# pause with a live agent pane must surface once because no run corroborates it. Here, +# fm-crew-state's run-step reconciliation has already corroborated the wait against the +# pipeline's own state (ci-monitoring-only, no gate, not fixing), so pause_state_class +# must trust it immediately and absorb on the bounded cadence even though the pane looks +# just as alive - a worker's own supervising loop staying alive to poll CI is expected. +test_run_step_paused_trusted_despite_live_agent_pane() { + local dir state fakebin out capture_file statusf window key pane_hash sig pid wakes + dir=$(make_case run-step-pause-live-agent); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; statusf="$state/ci.status" + window="test:fm-ci" + printf 'idle background ci monitoring\n' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=grok\nbackend=tmux\n' "$window" > "$state/ci.meta" + printf 'paused: waiting on ci to go green\n' > "$statusf" + sig=$(seen_sig "$statusf"); printf '%s' "$sig" > "$state/.seen-ci_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle background ci monitoring") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_FAKE_TMUX_CURRENT_COMMAND=grok \ + FM_FAKE_CREW_STATE='state: paused · source: run-step · declared wait (ci monitoring): waiting on ci to go green' \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_PAUSE_RESURFACE_SECS=999 FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid" + fail "a run-step-corroborated declared pause surfaced despite a live supervising agent pane: $(cat "$out")" + fi + [ -e "$state/.paused-$key" ] || { reap "$pid"; fail "run-step-corroborated pause did not enter the bounded pause cadence"; } + reap "$pid" + wakes=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w { n++ } END { print n + 0 }' "$state/.wake-queue" 2>/dev/null || echo 0) + [ "$wakes" -eq 0 ] || fail "run-step-corroborated pause with a live agent pane surfaced $wakes stale wakes instead of absorbing" + pass "a run-step-corroborated declared pause absorbs on the bounded cadence even with a live-looking supervising agent pane" +} + test_secondmate_paused_resurfaces_in_normal_mode() { local dir state fakebin out capture_file statusf window key pane_hash sig pid back dir=$(make_case secondmate-paused-resurface); state="$dir/state"; fakebin="$dir/fakebin" @@ -2941,6 +2978,7 @@ test_afk_busy_declared_pause_ticking_pane_hands_off_once test_nonterminal_stale_not_working_surfaced test_nonterminal_stale_paused_absorbed_then_resurfaced test_exited_declared_pause_is_bounded_but_live_gate_surfaces +test_run_step_paused_trusted_despite_live_agent_pane test_secondmate_paused_resurfaces_in_normal_mode test_secondmate_captain_held_resurfaces_in_normal_mode test_secondmate_nonpaused_stale_remains_suppressed From 4d984c17d2c0cf14c43fdac36f37e9a3adb02ed3 Mon Sep 17 00:00:00 2001 From: cm-maple7 Date: Mon, 31 Aug 2026 20:57:51 -0700 Subject: [PATCH 3/6] no-mistakes(review): Add test: declared pause beats checks-green ci reconciliation --- tests/fm-crew-state.test.sh | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 9821a632ca3..0d912465678 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -598,6 +598,28 @@ test_ci_monitoring_declared_pause_wins_over_working() { pass "a declared pause during ci-only monitoring wins over the working run-step reading" } +# The declared-wait override must also win once checks have already gone green: +# the checks-green branch above independently flips RUN_STATE to "done" while +# CI_STEP_STATUS stays "running", so the override must still fire on that state +# rather than being masked by the earlier done reclassification. This is the +# literal incident scenario: fm-crew-state reporting "done, still monitoring" +# and outranking the worker's declared paused: line. +test_ci_monitoring_declared_pause_wins_over_checks_green() { + reset_fakes + local d; d=$(new_case ci-green-paused) + make_repo_on_branch "$d/wt" fm/feat-cigreenpaused + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-cigreenpaused.meta" "window=fm:fm-feat-cigreenpaused" "worktree=$d/wt" "kind=ship" + printf 'paused: PR reserved for the captain to merge\n' > "$d/state/feat-cigreenpaused.status" + FM_FAKE_AXI_STATUS="$(run_ci_monitoring fm/feat-cigreenpaused)" + FM_FAKE_CI_LOGS="all CI checks passed - still monitoring until merged or closed" + local out; out=$(run_crew_state "$d" feat-cigreenpaused) + assert_contains "$out" "state: paused" "declared wait during green ci monitoring -> paused" + assert_contains "$out" "source: run-step" "declared wait during green ci monitoring -> run-step source" + assert_not_contains "$out" "state: done" "declared wait must not be masked by the checks-green done reclassification" + pass "a declared pause during ci monitoring wins even after checks have already gone green" +} + # The same declared pause must NOT win once the run has genuinely active work # left - a gate awaiting review, in this case - since the declaration there is # stale, not a legitimate description of the current wait. @@ -1766,6 +1788,7 @@ test_ci_monitoring_green_then_rearm_stays_working test_ci_monitoring_no_checks_yet_stays_working test_ci_monitoring_still_waiting_stays_working test_ci_monitoring_declared_pause_wins_over_working +test_ci_monitoring_declared_pause_wins_over_checks_green test_gated_run_declared_pause_does_not_override test_ci_monitoring_green_then_new_issue_stays_working test_ci_ready_done_log_relapse_stays_working From 5424decc8413c0fd085c554b3574837f2d2ce8d0 Mon Sep 17 00:00:00 2001 From: cm-maple7 Date: Mon, 31 Aug 2026 21:07:20 -0700 Subject: [PATCH 4/6] no-mistakes(review): Anchor crew_run_step_paused to source field, not substring --- bin/fm-classify-lib.sh | 11 ++++++----- tests/fm-watch-triage.test.sh | 25 +++++++++++++++++++++++++ 2 files changed, 31 insertions(+), 5 deletions(-) diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 91bdd5a139c..c3fc202db8f 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -1321,13 +1321,14 @@ crew_is_paused() { # # read (crew_absorb_class already made one), but only inside the already-rare # declared-pause branch its caller gates this behind, never per poll. crew_run_step_paused() { # - local id=$1 line + local id=$1 line state src [ -n "$id" ] || return 1 line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || return 1 - case "$line" in - "state: paused"*"source: run-step"*) return 0 ;; - *) return 1 ;; - esac + case "$line" in state:*) ;; *) return 1 ;; esac + state=${line#state: }; state=${state%% *} + [ "$state" = paused ] || return 1 + src=${line#*source: }; src=${src%% *} + [ "$src" = run-step ] } # Directories excluded from the worktree write probe below, and the depth it walks. diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index 8140849120d..2fe533a1a41 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -377,6 +377,30 @@ test_crew_absorb_class_classifier() { pass "crew_absorb_class: working/paused/none from one read; crew_is_paused and crew_is_provably_working agree" } +# crew_run_step_paused: the run-step-vs-status-log distinction pause_state_class +# trusts to skip the live-agent-liveness gate. Must anchor to the actual source +# field (fm-crew-state.sh's fixed "state: X · source: Y · detail" format) rather +# than searching the whole line for the substring "source: run-step" - a worker's +# own free-text status note (the detail field, on the status-log fallback path) +# can legitimately contain that phrase without the pause being run-step-attributed. +test_crew_run_step_paused_classifier() { + local dir fakebin + dir=$(make_case run-step-paused-classifier); fakebin="$dir/fakebin" + export FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" + export FM_FAKE_CREW_STATE + FM_FAKE_CREW_STATE='state: paused · source: run-step · declared wait (ci monitoring): waiting on ci to go green' + crew_run_step_paused a || fail "run-step-sourced pause not recognized" + FM_FAKE_CREW_STATE='state: paused · source: status-log · waiting on source: run-step to finish' + ! crew_run_step_paused a || fail "status-log pause with a free-text mention of 'source: run-step' was misclassified as run-step-attributed" + FM_FAKE_CREW_STATE='state: paused · source: status-log · awaiting upstream' + ! crew_run_step_paused a || fail "plain status-log pause treated as run-step-attributed" + FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + ! crew_run_step_paused a || fail "a working run-step verdict treated as paused" + ! crew_run_step_paused "" || fail "empty id treated as run-step-paused" + unset FM_FAKE_CREW_STATE + pass "crew_run_step_paused anchors to the source field, never a free-text substring match" +} + # The wedge detector's third liveness input: writes inside the crew's own recorded # worktree. Every negative outcome must report "no evidence" so the caller keeps # its existing escalation schedule, and a supervisor-side git read (which touches @@ -2948,6 +2972,7 @@ test_classifier_primitives test_crew_is_provably_working_classifier test_status_is_paused_classifier test_crew_absorb_class_classifier +test_crew_run_step_paused_classifier test_crew_worktree_written_since_classifier test_empty_write_prune_widens_the_probe test_empty_write_prune_from_the_environment_widens_the_probe From a664996fc713f5b9e23576cc7b2bcdb3e7bd4a9a Mon Sep 17 00:00:00 2001 From: cm-maple7 Date: Mon, 31 Aug 2026 21:20:16 -0700 Subject: [PATCH 5/6] no-mistakes(document): Update architecture.md for wedge re-verify and pause-trust changes --- docs/architecture.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/architecture.md b/docs/architecture.md index 167b53468c5..70a23c65722 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -15,6 +15,8 @@ A pane holding a file newer than the start of its own quiet window, anywhere in That deferral re-surfaces on the same `FM_PAUSE_RESURFACE_SECS` cadence as a declared wait, with a reason naming the write evidence rather than a wedge, and it is bounded to one pruned, depth-bounded, wall-clock-bounded walk (`FM_WORKTREE_WRITE_PRUNE`, `FM_WORKTREE_WRITE_MAXDEPTH`, `FM_WORKTREE_WRITE_TIMEOUT`) taken only in the branch that was about to escalate, never on every poll. Every absence of write evidence, including a missing worktree record, a torn-down worktree, a walk that outlives its wall-clock bound on a hung mount, and a failed walk, leaves the existing escalation schedule untouched, so a crew that writes nothing still escalates exactly as before. A secondmate's recorded worktree is never probed for write activity, because it is a provisioned firstmate home whose own supervision keeps writing inside it whether or not the mate produces anything, so its panes keep escalating on the unchanged schedule. +A provably-working escalation whose working verdict came from a run-step reading, rather than a busy pane, is re-verified against a fresh run-step read at that same threshold before it actually escalates - the same never-escalate-past-positive-evidence contract as the worktree-write probe, applied to a background run's own authoritative reading instead of a file timestamp - so a no-mistakes validation that legitimately sits on an idle pane far longer than `FM_STALE_ESCALATE_SECS` keeps absorbing and resetting its timer for as long as the run-step still reports working; only once it stops does the escalation proceed. +That reverification is scoped to the run-step-classified wedge chain alone and leaves the separate busy-pane/no-completed-turn wedge bound (`FM_BUSY_TURN_MAX_SECS`) unchanged. A busy pane is otherwise exempt from staleness, but only until its latest `state/.turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, worktree-write deferral, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. A crew that declared an external wait (`paused:`) or a verified captain-held transfer is the one exception to that bound: its busy verdict supplies liveness while identifying the long-running foreground call as the declared wait, so it takes the bounded `FM_PAUSE_RESURFACE_SECS` recheck instead of a wedge escalation. Lifting the declaration restores the unchanged busy-pane wedge path, while a pane that is no longer busy returns to the existing idle declared-wait classification. @@ -35,6 +37,8 @@ No-verb wakes, such as `working:` notes and bare turn-ended signals, are benign A `kind=secondmate` task's status signal is the parent-directed reply stream and is never absorbed as provably working; only its bare turn-ended signal retains the ordinary absorb rule. A crew that declares `paused:` for a known external wait, or carries a verified `captain-held` transfer, is separately absorbed while idle and re-surfaced only on the longer pause cadence, rather than being treated as a possible wedge. For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint only when the backend confidently reports its agent dead. +A pause corroborated by an authoritative run-step reading - the run's only remaining step is CI monitoring, no gate, not fixing - skips that dead-agent gate entirely and is trusted immediately, since a worker's own supervising loop staying alive to poll CI is expected, not evidence of staleness; that same run-step reconciliation reports paused instead of the run-step's own working/done reading, so a run parked purely on CI monitoring can no longer silently outrank its own declared wait and wedge-escalate forever. +A pause backed only by the status log, with no run to corroborate it, still needs the dead-agent gate, because a live-looking agent there may be sitting at an undeclared interactive gate rather than the wait it named. Live or inconclusive liveness remains fail-open at that initial surface, and a secondmate's endpoint liveness is still never read at all; a mate is admitted to that same cadence only to serve a declared wait's bounded re-surface, so a forgotten pause or captain hold on a mate cannot rot invisibly. Its initial normal-mode status signal still surfaces through the no-verb path, while away mode self-handles that routine signal and owns the later recheck. Fresh stale panes use the same current-state read before trusting the status log, so an active run or a proven busy worker outranks an old captain-relevant status-log line left behind before validation. From 3559bfc422a1deab57b8e383b578b7bc10d2661a Mon Sep 17 00:00:00 2001 From: cm-maple7 Date: Mon, 31 Aug 2026 23:26:32 -0700 Subject: [PATCH 6/6] no-mistakes(document): Sync stale header comments in fm-watch.sh/fm-classify-lib.sh with new reverify/pause logic --- bin/fm-classify-lib.sh | 5 ++++- bin/fm-watch.sh | 8 ++++++-- 2 files changed, 10 insertions(+), 3 deletions(-) diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index c3fc202db8f..d5d572dcdad 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -19,7 +19,10 @@ # to decide whether a crew that just stopped its turn or went stale is working, # deliberately paused, or neither. Callers run it ONLY on no-verb signal handling # and first sighting of a stale hash, never on every wake, so the per-wake triage -# stays cheap. status_open_decisions_incremental (see "incremental (cursor-backed) +# stays cheap. crew_run_step_paused makes that same bounded fm-crew-state.sh call +# a second time, but only inside the already-rare declared-pause branch its +# caller gates it behind, to tell a run-step-corroborated pause from a +# status-log-only one. status_open_decisions_incremental (see "incremental (cursor-backed) # open-decisions fold" below) also writes: it persists a per-status-file byte # cursor and folded open-set as a side effect, so a per-drain fleet-wide scan # stays bounded by new appends instead of re-reading each task's whole lifetime diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 2d69fdaf17a..4307417f519 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -27,8 +27,12 @@ # applies does the log's last line decide: # terminal (captain-relevant) or non-terminal (no verb), # both surfaced at once. A provably-working stale past the -# wedge threshold also surfaces, with an "escalation N" -# count in the reason; at FM_WEDGE_DEMAND_INSPECT_COUNT +# wedge threshold surfaces with an "escalation N" count in +# the reason, EXCEPT a run-step-classified working chain +# (a background no-mistakes validation, not a busy pane), +# which re-verifies the run-step at that same threshold +# and keeps absorbing for as long as it still reports +# working. At FM_WEDGE_DEMAND_INSPECT_COUNT # consecutive escalations on the SAME pane, the reason # also carries a "demand-deep-inspection" marker so the # wake payload itself, not just repetition, forces a