diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 8bd4fe746ad..246df9e2b46 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -1795,6 +1795,35 @@ crew_is_paused() { # [ "$(crew_absorb_class "$1")" = paused ] } +# The note bin/fm-crew-state.sh appends to an active run-step's detail when the +# pipeline's own recency verdict says that step is still producing activity. One +# definition, written by fm-crew-state.sh and matched by the predicate below, so +# the emitted line and the classifier reading it cannot drift apart. +FM_CREW_STATE_ACTIVITY_RECENT='run activity recent' + +# 0 only on POSITIVE proof that crew 's OWN attributed no-mistakes run is +# still doing work: fm-crew-state.sh reports a working run-step for THIS crew and +# marks its active step's activity recent, which is the pipeline's own recency +# verdict (`axi status` prefixes last_activity with `quiet` once nothing has +# arrived), never a second threshold invented here and never the liveness of the +# shared daemon, which any other crew's run keeps up. That distinction is the +# whole point: a record left at running/fixing after a drive call was killed, or +# after the daemon exited under it, reports a working run-step while nothing +# executes it, and must NOT read as work in progress. +# The busy-pane half of crew_absorb_class's `working` is deliberately excluded: a +# caller that already holds a busy verdict cannot let that pane vouch for itself. +# Not a pure read (see crew_absorb_class), so callers run it at most once per +# STALE_ESCALATE_SECS - never per poll. +crew_nm_run_activity_is_recent() { # + local id=$1 line + [ -n "$id" ] || return 1 + line=$("$FM_CREW_STATE_BIN" "$id" 2>/dev/null) || true + case "$line" in + "state: working"*"source: run-step"*"$FM_CREW_STATE_ACTIVITY_RECENT"*) return 0 ;; + esac + return 1 +} + # Directories excluded from the worktree write probe below, and the depth it walks. # The excluded set is everything a supervisor read or a package manager can write # without the crew doing any work - .git first, so firstmate's own read-only git diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index f3a99c3e3e5..cb7f78af38d 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -61,6 +61,12 @@ # FAILED record whose daemon an explicit probe proves down reads unknown, # never failed: an instrument failure must not read as work failure # (nm_daemon_probe_down). +# A working run-step also carries a positive `run activity recent` note in +# its detail while the pipeline's own recency verdict says an active step +# is still reporting (nm_run_activity_is_recent, which requires the +# captured `active_steps[]` table and never treats an absent one as +# recency). Supervisors read that note to tell an advancing run from a +# record nothing is executing. # 3. Reconcile the status log: if its last line says needs-decision/blocked but # the run-step shows the run moved on, the log is deterministically stale and # is flagged superseded. A genuinely parked run plus a needs-decision log @@ -743,6 +749,14 @@ if [ "$HAVE_RUN" = 1 ]; then ;; esac + # Positive recency, for supervisors that must tell an advancing run from a + # record nothing is executing: the client's own `quiet` prefix is the verdict + # (nm_run_activity_is_recent), so the note appears only while an active step + # keeps reporting, and never for a coarse row with no steps table to read. + if [ "$RUN_STATE" = working ] && nm_run_activity_is_recent; then + RUN_DETAIL="$RUN_DETAIL${SEP}$FM_CREW_STATE_ACTIVITY_RECENT" + fi + emit "$RUN_STATE" run-step "$RUN_DETAIL" fi diff --git a/bin/fm-dod-lib.sh b/bin/fm-dod-lib.sh index 07a7b46e242..312b238cc5e 100755 --- a/bin/fm-dod-lib.sh +++ b/bin/fm-dod-lib.sh @@ -198,9 +198,13 @@ fm_dod_block() { # # Definition of done Delivery contract: mode=direct-PR This task ships **direct-PR**: you raise the PR yourself, without the no-mistakes pipeline. +Do not run /no-mistakes unless firstmate explicitly instructs you to change this task's delivery path. The task is complete only when committed on your branch. -When it is implemented and committed, push your branch and open a PR with \`gh-axi\`, then append \`done: PR {url}\` to the status file and stop. -Do NOT run /no-mistakes. The configured merge authority decides whether to merge the PR; firstmate relays the outcome. +When it is implemented and committed, push your branch and open a PR with \`gh-axi\`. +If a push, PR creation, or PR verification fails, diagnose the forge failure first, including the reported authentication, remote, branch, or API error; do not use no-mistakes as a workaround. +Before the final status, verify with \`gh-axi\` that the branch was actually pushed and that the forge reports a full \`https://...\` PR URL for that branch. +Only after those checks append \`done: PR {url}\` to the status file and stop. +The configured merge authority decides whether to merge the PR; firstmate relays the outcome. EOF ;; local-only) @@ -237,6 +241,9 @@ So background the drive call and poll \`no-mistakes axi status\` from a separate Where a harness's own command limit is not established, assume it bounds commands and use that same background-and-poll shape. A killed or timed-out call is never evidence the daemon died: the daemon accepts your response immediately and runs the round in the background, so the call was only ever waiting for a read while the run kept working. Reattach and keep going rather than reporting the pipeline blocked; rule 7 owns the checks that decide when a pipeline block is real. +After every \`no-mistakes axi respond\`, continue in the same turn with bounded calls to the structured \`no-mistakes axi status\` interface until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block. +An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run. +Never end your turn or promise to resume or check later while structured status shows that validation is active, unless the attributed run presents a genuine ask-user decision - escalate it and stop - or rule 7's daemon checks have established a real block. Two firstmate-specific rules layer on top of that guidance: - ask-user findings are never yours to answer: escalate to firstmate using rule 6's ask-user format and stop. diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index f4246480ff5..1932c6de0eb 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -214,8 +214,10 @@ STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provabl # non-busy stale - so it escalates via the existing stale reason, escalation # counter, and demand-deep-inspection marker for human inspection only, never an # automatic interrupt, signal, or restart - unless the crew declared the wait -# itself, which takes the long pause cadence instead. A completed turn touches -# turn-ended and resets the age. Set generously above any legitimate interval +# itself, which takes the long pause cadence instead, or the crew's own +# no-mistakes run reports recent activity at the escalation threshold, which +# defers that one escalation and must prove itself again for the next. +# A completed turn touches turn-ended and resets the age. Set generously above any legitimate interval # between completed turns, including long tool calls, builds, or test runs. BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # A local secondmate's foreign queue is checked on every poll, but only after this @@ -853,9 +855,16 @@ clear_write_tracking() { # # line that an active run/busy pane outranked). # The worktree write probe runs ONLY here, inside the at-threshold branch that is # about to escalate: at most one bounded walk per window per STALE_ESCALATE_SECS, -# never per poll. -wedge_timer_check() { # - local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 since age n reason +# never per poll. `defer-live-run` adds the one other evidence a pane can offer +# there, on the same terms and for the same reason: this crew's OWN no-mistakes +# run reporting recent activity defers the escalation once, restarting the idle +# timer so the next window must prove it again. The moment that run stops +# reporting recent activity - a hung step, a record left behind after the daemon +# exited, a run that is no longer this crew's - the threshold falls straight +# through to the unchanged escalation ladder and its demand-deep-inspection +# marker, so no deferral can repeat without fresh proof of work. +wedge_timer_check() { # [defer-live-run] + local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 live_run=${6-} since age n reason since=$(cat "$since_file" 2>/dev/null || true) case "$since" in ''|*[!0-9]*) @@ -872,6 +881,11 @@ wedge_timer_check() { # "$since_file" + triage_log "absorbed $label (this crew's own no-mistakes run reports recent activity, idle ${age}s): $win" + return 0 + fi n=$(( $(cat "$escalation_file" 2>/dev/null || echo 0) + 1 )) echo "$n" > "$escalation_file" reason="stale: $win (idle ${age}s, possible wedge, escalation $n)" @@ -947,9 +961,16 @@ handle_paused_stale() { # # A busy pane past BUSY_TURN_MAX_SECS is normally a wedge suspect because a hung # foreground call can hide behind a busy signature. A `paused:` declaration or # verified captain-held transfer instead identifies that live foreground call as -# the expected external wait. The caller has already confirmed liveness through -# the busy verdict, so this exception does not suppress undeclared wedges or -# alter the separate non-busy classification. handle_paused_stale keeps the +# the expected external wait. An advancing no-mistakes validation is the other +# real wait - a validating worker holds ONE turn open for the whole run by +# contract, so its completed-turn age is expected to cross the bound - but that +# evidence is NOT read here: it is handed to wedge_timer_check as +# `defer-live-run`, which consults it only in the branch that is about to +# escalate. That keeps the crew-state read to at most one per window per +# STALE_ESCALATE_SECS instead of one per poll, and keeps the outcome a single +# deferral that must be re-earned rather than a standing exemption. The caller +# has already confirmed liveness through the busy verdict, so this exception does +# not suppress undeclared wedges or alter the separate non-busy classification. handle_paused_stale keeps the # exception bounded by re-surfacing it once per PAUSE_RESURFACE_SECS. Away mode # remains daemon-owned and receives the undecorated wake identity for its own # classification, which is why the declaration is read before the afk branch @@ -992,7 +1013,7 @@ busy_turn_bound_check() { # .turn-ended` marker reaches `FM_BUSY_TURN_MAX_SECS`, or its `state/.meta` spawn record reaches that age before any turn completes; past that bound it is routed through the same wedge escalation, with the identical reason, escalation count, worktree-write deferral, and `demand-deep-inspection` marker, for inspection only - never an automatic interrupt, signal, or restart. -A crew that declared an external wait (`paused:`) or a verified captain-held transfer is the one exception to that bound: its busy verdict supplies liveness while identifying the long-running foreground call as the declared wait, so it takes the bounded `FM_PAUSE_RESURFACE_SECS` recheck instead of a wedge escalation. +A crew that declared an external wait (`paused:`) or a verified captain-held transfer is one exception to that bound: its busy verdict supplies liveness while identifying the long-running foreground call as the declared wait, so it takes the bounded `FM_PAUSE_RESURFACE_SECS` recheck instead of a wedge escalation. Lifting the declaration restores the unchanged busy-pane wedge path, while a pane that is no longer busy returns to the existing idle declared-wait classification. -While away mode is active, a busy pane that crosses the bound under a declared wait is handed to the daemon as the plain wake identity instead of taking that recheck in the watcher, because the daemon owns triage there and a wake already decorated as a possible wedge would override the daemon's own declared-wait verdict; an undeclared busy pane past the bound still takes the wedge escalation in away mode. +The other exception is a worker inside its own no-mistakes validation, whose contract holds one turn open for the whole run so its completed-turn age is expected to cross the bound: at each `FM_STALE_ESCALATE_SECS` threshold, a busy pane whose own attributed run reports recent activity defers that one escalation and restarts the idle timer, so the next threshold has to prove the activity again. +The proof is the pipeline's own recency verdict on the step attributed to this crew, read through `bin/fm-crew-state.sh` only in the branch that was about to escalate rather than on every poll, and never the liveness of the shared no-mistakes daemon, which any other crew's run keeps up. +A hung step, a record left behind after the daemon exited under it, and a run that is no longer this crew's all stop reporting that activity, so the threshold falls straight through to the unchanged escalation ladder and its `demand-deep-inspection` marker; this deferral exists only on the busy-pane path, and the non-busy stale classification is unchanged. +While away mode is active, a busy pane that crosses the bound under a declared wait is handed to the daemon as the plain wake identity instead of taking that recheck in the watcher, because the daemon owns triage there and a wake already decorated as a possible wedge would override the daemon's own declared-wait verdict; an undeclared busy pane past the bound still takes the wedge path in away mode, including the evidence-based deferrals above. That handoff is keyed on the declaration itself (the status log's signature) rather than on the pane capture, so a harness footer that ticks on every poll wakes the daemon once per declaration instead of once per poll, and it clears the wedge timer, escalation count, and worktree-write deferral exactly as the normal-mode absorber does, so an undeclared busy phase's timer does not resume when the declaration lifts. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) only after generation-bound recovery evidence is published, so an interrupted watcher or handling turn can be recovered without losing the queue record. Agent endpoint liveness and queue-consumption liveness are separate: on each poll, the primary watcher reads the oldest valid actionable row from every endpoint-recorded local secondmate home's durable wake queue without locking, consuming, or rewriting that foreign queue. diff --git a/docs/configuration.md b/docs/configuration.md index ffe847b7f10..868daa9fee8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -931,7 +931,7 @@ FM_TURNEND_CHURN_ABSORB_SECS=900 # longest one endpoint's bare turn-ends may b FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless admitted directly to the declared-wait cadence, while a live idle declared wait still surfaces once before that cadence bounds repeats -FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/.turn-ended marker, or its state/.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait or verified captain-held transfer takes the FM_PAUSE_RESURFACE_SECS recheck below instead +FM_BUSY_TURN_MAX_SECS=3600 # busy-turn age bound in seconds; docs/architecture.md "Event-driven supervision" owns the age sources, inspection-only escalation, and declared-wait and recent-run-activity deferrals FM_PAUSE_RESURFACE_SECS=3600 # seconds between bounded rechecks of a declared external wait or verified captain-held transfer, and between repeated new-hash stale alarms for an ordinary crew task with an open backlog captain call; this includes a live idle pane after its first inconclusive stale wake and a live busy pane past FM_BUSY_TURN_MAX_SECS, while the away-mode daemon uses the same setting and ages its window against the crew's own latest status line rather than pane busy state FM_SECONDMATE_WAKE_STALL_SECS=180 # minimum interval with no change of the oldest actionable foreign wake-queue row (it advances as the mate drains, and a queue reprovisioned under the same task id starts a fresh interval at whatever sequence it restarts) before an endpoint-recorded local secondmate produces one durable parent wake-loop-stall notification for that no-progress episode; a mate that is provably inside an active turn (an exact busy verdict, bounded by the same FM_BUSY_TURN_MAX_SECS above) never escalates whatever this interval says, declared external-wait pause rows are excluded, and zero or invalid values use 180 FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index f8e63fa8ca4..1cbf3c4575c 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -818,7 +818,10 @@ The same guarded named-lab command passed on 2026-09-03 against Herdr 0.8.2 afte It reported `steal_live=0 floor_verdict=0 default-session-tripwire=armed`, with the fleet's default session unchanged before and after. Part C is the case the suite could not reach before: a doomed pane whose shell holds a persistent background child fails the lone-idle-shell proof on every sample, so the plan takes the plain explicit close, in the geometry where the closing workspace's right neighbour is a spacer rather than the focused anchor. -On 0.7.5 that fallback exposed a bounded four-sample wrong-focus window and restored the anchor exactly; on 0.8.0 the same fallback exposed none, which is why default-on projection is floored at 0.8.0 rather than mitigated further below it. +In the recorded runs above, the sampler observed a bounded four-sample wrong-focus window on 0.7.5 with exact anchor restoration and none on 0.8.0, supporting the 0.8.0 default-on floor. +The current Part C regression instead checks the adapter's call log for corrective `tab focus` and verifies the final exact anchor: correction must occur on a release Part A proves defective and must be absent on a focus-preserving release. +It no longer samples focus concurrently during the fallback close; the recorded sample counts are prior evidence, not output of the current guard. +Part B retains concurrent sampling of the mitigated removal path. The suite also cross-checks its own Part A measurement against the floor classifier on whatever release it runs, so a drifted protocol-to-release mapping fails there rather than silently gating on the wrong thing. ### Presentation version floor diff --git a/tests/fm-backend-herdr-focus-flash-e2e.test.sh b/tests/fm-backend-herdr-focus-flash-e2e.test.sh index 69ec69e8224..0a8a7206f4f 100755 --- a/tests/fm-backend-herdr-focus-flash-e2e.test.sh +++ b/tests/fm-backend-herdr-focus-flash-e2e.test.sh @@ -266,33 +266,12 @@ while [ "$C_CHILD_ATTEMPT" -lt 100 ]; do done [ "$C_CHILD_STABLE" -ge 2 ] || fail 'the Part C doomed pane never reported a stable persistent child process' +# The plain close's wrong-focus window is bounded by the operation itself, so +# it is read from the call log rather than raced with an external sampler: the +# corrective `tab focus` is issued exactly when the adapter's own post-close +# snapshot differs from the pre-operation one. C_CALL_LOG="$TMP_ROOT/call-c.log" -C_FOCUS_SAMPLES="$TMP_ROOT/focus-c.samples" -C_OPERATION_ACTIVE="$TMP_ROOT/operation-c.active" -C_SAMPLER_READY="$TMP_ROOT/sampler-c.ready" -SAMPLER_STOP="$TMP_ROOT/sampler-c.stop" : > "$C_CALL_LOG" -: > "$C_FOCUS_SAMPLES" -( - : > "$C_SAMPLER_READY" - while [ ! -e "$SAMPLER_STOP" ]; do - if [ -e "$C_OPERATION_ACTIVE" ]; then - if C_SAMPLE=$(focus_snapshot); then - printf '%s\n' "$C_SAMPLE" >> "$C_FOCUS_SAMPLES" - else - printf '%s\n' UNREADABLE >> "$C_FOCUS_SAMPLES" - fi - fi - done -) & -SAMPLER_PID=$! -C_READY_ATTEMPT=0 -while [ ! -e "$C_SAMPLER_READY" ] && [ "$C_READY_ATTEMPT" -lt 100 ]; do - sleep 0.01 - C_READY_ATTEMPT=$((C_READY_ATTEMPT + 1)) -done -[ -e "$C_SAMPLER_READY" ] || fail 'the Part C focus sampler did not start' -: > "$C_OPERATION_ACTIVE" # A short proof budget keeps the exhausted-proof path fast; the count below is # what proves the proof was exhausted rather than skipped. C_PROOF_POLLS=3 @@ -308,10 +287,6 @@ C_OUT=$(PATH="$FAKEBIN:$HERDR_ORIGINAL_PATH" FM_FLASH_CALL_LOG="$C_CALL_LOG" \ fm_backend_herdr_projection_close_pane_focus_preserving "$2" "$3" ' _ "$ROOT" "$HERDR_LAB_SESSION" "$C_DOOMED_PANE" 2>&1) C_STATUS=$? -rm -f "$C_OPERATION_ACTIVE" -: > "$SAMPLER_STOP" -wait "$SAMPLER_PID" 2>/dev/null || true -SAMPLER_PID= [ "$C_STATUS" -eq 0 ] || fail "the production focus-preserving close failed (status $C_STATUS): $C_OUT" wait_ws_gone "$C_DOOMED_WS" || fail 'the fallback close left the doomed workspace behind' if lab pane get "$C_DOOMED_PANE" >/dev/null 2>&1; then @@ -333,19 +308,18 @@ pass 'fallback: a doomed pane holding a persistent child exhausts the proof and C_AFTER=$(focus_snapshot) || fail 'could not capture the Part C post-close focus' [ "$C_AFTER" = "$C_BEFORE" ] \ || fail "the fallback close left focus off the anchor ($C_BEFORE -> $C_AFTER)" -C_WRONG=$(grep -Fvxc -- "$C_BEFORE" "$C_FOCUS_SAMPLES" || true) if [ "$STEAL_LIVE" = 1 ]; then # A defective release cannot make this path focus-safe, which is precisely why # default-on projection is floored above it. The wrong-focus window is # explicitly accepted here, but only as a BOUNDED one: the restore backstop - # must have put the anchor back exactly, and the whole exposure must end with + # must have fired and put the anchor back exactly, so the exposure ends with # the operation rather than parking the captain somewhere else. - [ "$C_WRONG" -ge 1 ] \ - || fail 'Part C reached the fallback on a defective release but observed no wrong-focus sample at all, so the sampler proved nothing' - pass "fallback on a defective release: a bounded wrong-focus window of $C_WRONG samples was fully restored to the anchor" + grep -q '^tab focus' "$C_CALL_LOG" \ + || fail 'Part C took the fallback on a defective release without the corrective restore, so no bounded wrong-focus window was exercised' + pass 'fallback on a defective release: the wrong-focus window the plain close opens was closed by the exact-tab restore' else - [ "$C_WRONG" -eq 0 ] \ - || fail "a focus-preserving release exposed $C_WRONG wrong-focus samples on the fallback path" + grep -q '^tab focus' "$C_CALL_LOG" \ + && fail 'a focus-preserving release needed the corrective restore, so the fallback path exposed a wrong-focus window' pass 'fallback on a focus-preserving release: the plain explicit close preserved exact focus throughout' fi diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 5467b5cbdca..1c3a8277627 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -372,6 +372,60 @@ test_no_mistakes_dod_wording() { pass "fm-brief.sh: no-mistakes DOD keeps its apostrophe prose and bans --yes outright" } +test_active_no_mistakes_validation_cannot_be_deferred() { + local home id brief + home="$TMP_ROOT/active-validation-home" + mkdir -p "$home/data" + id="brief-active-validation-c1" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --mode no-mistakes >/dev/null 2>&1 + brief="$home/data/$id/brief.md" + + assert_grep "After every \`no-mistakes axi respond\`, continue in the same turn" "$brief" \ + "no-mistakes brief did not require same-turn continuation after a gate response" + assert_grep "bounded calls to the structured \`no-mistakes axi status\` interface" "$brief" \ + "no-mistakes brief did not require bounded structured status polling" + assert_grep "until the attributed run changes step, reaches a terminal outcome, presents a genuine ask-user decision, or rule 7's daemon checks establish a real block" "$brief" \ + "no-mistakes brief did not define the only status-polling stop conditions" + assert_grep "An accepted response or a status that still reports active work is not a stopping point; the same continuation rule applies after starting or reattaching to a run." "$brief" \ + "no-mistakes brief did not extend the continuation rule past a gate response to starting and reattaching" + assert_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active, unless the attributed run presents a genuine ask-user decision - escalate it and stop - or rule 7's daemon checks have established a real block." "$brief" \ + "no-mistakes brief still permits deferring an active validation run" + pass "fm-brief.sh: active no-mistakes validation continues in the same turn through the next real transition" +} + +test_direct_pr_requires_forge_proof_and_diagnosis() { + local home id brief scout charter local_brief + home="$TMP_ROOT/direct-pr-proof-home" + mkdir -p "$home/data" + id="brief-direct-pr-proof-c2" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --mode direct-PR >/dev/null 2>&1 + brief="$home/data/$id/brief.md" + + assert_grep "Do not run /no-mistakes unless firstmate explicitly instructs you to change this task's delivery path." "$brief" \ + "direct-PR brief did not forbid an unrequested no-mistakes run" + assert_grep "diagnose the forge failure first" "$brief" \ + "direct-PR brief did not require forge-first diagnosis" + assert_grep "verify with \`gh-axi\` that the branch was actually pushed" "$brief" \ + "direct-PR brief did not require proof of the remote branch" + assert_grep "full \`https://...\` PR URL" "$brief" \ + "direct-PR brief did not require a verified full PR URL" + + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" unaffected-scout some-proj --scout >/dev/null 2>&1 + scout="$home/data/unaffected-scout/brief.md" + FM_HOME="$home" FM_SECONDMATE_CHARTER='Supervise assigned work.' \ + "$ROOT/bin/fm-brief.sh" unaffected-charter --secondmate --no-projects >/dev/null 2>&1 + charter="$home/data/unaffected-charter/brief.md" + FM_HOME="$home" "$ROOT/bin/fm-brief.sh" unaffected-local some-proj --mode local-only >/dev/null 2>&1 + local_brief="$home/data/unaffected-local/brief.md" + for unaffected in "$scout" "$charter" "$local_brief"; do + assert_no_grep "diagnose the forge failure first" "$unaffected" \ + "an unaffected scaffold received the direct-PR forge contract" + assert_no_grep "Never end your turn or promise to resume or check later while structured status shows that validation is active, unless the attributed run presents a genuine ask-user decision - escalate it and stop - or rule 7's daemon checks have established a real block." "$unaffected" \ + "an unaffected scaffold received the no-mistakes active-run contract" + done + pass "fm-brief.sh: direct-PR completion requires forge diagnosis, a pushed branch, and a verified full URL" +} + test_ask_user_escalation_format() { local home id brief mode other_id other_brief home="$TMP_ROOT/ask-user-home" @@ -878,6 +932,8 @@ test_ship_mode_is_explicit_not_registry test_delivery_flags_are_refused_where_they_do_not_apply test_faster_paths_use_configured_authority_without_stacked_review test_no_mistakes_dod_wording +test_active_no_mistakes_validation_cannot_be_deferred +test_direct_pr_requires_forge_proof_and_diagnosis test_ask_user_escalation_format test_ship_project_memory_wording test_herdr_lab_contract_is_explicit_and_complete diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 309a7008f8a..4620030e6bd 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -567,6 +567,41 @@ test_daemon_claim_over_live_run_reads_run_alive() { pass "daemon/timeout blocked claim over a live fixing run reads as run alive" } +# The emitted line is what supervisors classify from, so an ACTIVE run's own +# recency verdict has to reach it. `axi status` leaves last_activity unprefixed +# while a step keeps reporting and prefixes it with `quiet` once nothing arrives; +# only the first case earns the recency note, so a record still reading fixing +# while nothing executes it is distinguishable from a run doing work. +test_active_run_reports_its_own_activity_recency() { + reset_fakes + local d; d=$(new_case activity-recency) + make_repo_on_branch "$d/wt" fm/feat-ar + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-ar.meta" "window=fm:fm-feat-ar" "worktree=$d/wt" "kind=ship" + + FM_FAKE_AXI_STATUS="$(run_fixing_active_recent fm/feat-ar)" + local out; out=$(run_crew_state "$d" feat-ar) + assert_contains "$out" "state: working" "a reporting fixing run is working" + assert_contains "$out" "run activity recent" "a reporting active step earns the recency note" + + FM_FAKE_AXI_STATUS="$(run_fixing_active_quiet fm/feat-ar)" + out=$(run_crew_state "$d" feat-ar) + assert_contains "$out" "source: run-step" "a quiet fixing run is still run-step sourced" + assert_not_contains "$out" "run activity recent" \ + "a quiet active step must not read as a run doing work" + + # The coarse ledger fallback has no steps table to read, so it can never claim + # recency: absent positive evidence, the note stays off. + local short; short=$(git -C "$d/wt" rev-parse --short=7 HEAD) + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST=" running fm/feat-ar ${short} 2026-07-02 22:05" + out=$(run_crew_state "$d" feat-ar) + assert_contains "$out" "state: working" "the coarse row still attributes this branch's run" + assert_not_contains "$out" "run activity recent" \ + "the coarse runs-list fallback cannot claim activity recency" + pass "an active run-step reports the pipeline's own activity recency, and only on positive evidence" +} + # A genuine refused socket outranks the persisted fixing record, which can # survive after the daemon exits. test_socket_refusal_over_stale_fixing_run_reports_blocked() { @@ -2235,6 +2270,7 @@ test_active_run_is_authoritative test_stale_needs_decision_superseded test_stale_blocked_superseded test_daemon_claim_over_live_run_reads_run_alive +test_active_run_reports_its_own_activity_recency test_socket_refusal_over_stale_fixing_run_reports_blocked test_socket_refusal_over_terminal_run_reports_blocked test_ordinary_blocked_over_live_run_keeps_plain_superseded diff --git a/tests/fm-task-delivery.test.sh b/tests/fm-task-delivery.test.sh index bdace4b501e..9139ce9ed84 100755 --- a/tests/fm-task-delivery.test.sh +++ b/tests/fm-task-delivery.test.sh @@ -380,6 +380,8 @@ STUB "promoted no-mistakes worker did not receive the --yes prohibition" assert_grep "It is banned fleet-wide" "$payload" \ "promoted no-mistakes worker did not receive the fleet-wide ban wording" + assert_grep "After every \`no-mistakes axi respond\`, continue in the same turn" "$payload" \ + "promoted no-mistakes worker can still defer an active validation run" payload="$TMP_ROOT/promote-dod/payload-promote-dod-direct-pr" assert_grep "supersede the scout delivery rules and report-based Definition of done" "$payload" \ @@ -388,8 +390,12 @@ STUB "promoted worker lost the scout protocols and safety rules that still apply" # The faster paths keep their own contracts rather than inheriting the pipeline's. - assert_grep "Do NOT run /no-mistakes" "$payload" \ + assert_grep "Do not run /no-mistakes unless firstmate explicitly instructs you to change this task's delivery path." "$payload" \ "promoted direct-PR worker lost its no-pipeline contract" + assert_grep "verify with \`gh-axi\` that the branch was actually pushed" "$payload" \ + "promoted direct-PR worker lost its pushed-branch proof" + assert_grep "diagnose the forge failure first" "$payload" \ + "promoted direct-PR worker lost its forge-first diagnosis contract" assert_grep "Do NOT push, do NOT open a PR, do NOT merge" "$TMP_ROOT/promote-dod/payload-promote-dod-local-only" \ "promoted local-only worker lost its no-remote contract" assert_no_grep "no-mistakes axi respond" "$TMP_ROOT/promote-dod/payload-promote-dod-direct-pr" \ diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index 029b4a491f9..76c74998388 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -3176,6 +3176,82 @@ test_busy_pane_repeated_escalation_reaches_demand_deep_inspection() { pass "repeated busy turn-age escalations reuse the existing escalation counter and demand deep inspection at the threshold" } +# --- live no-mistakes run + busy pane: the run IS the declared wait ---------- +# A worker driving no-mistakes holds ONE turn open for the whole validation by its +# generated Definition of done (bin/fm-dod-lib.sh), and a run chains fix rounds +# well past BUSY_TURN_MAX_SECS, so a healthy validating crew crosses the +# completed-turn bound as a matter of course. Its escalation is DEFERRED, never +# cancelled, and only on the pipeline's own recency verdict for THIS crew's run: +# `axi status` prefixes an active step's last_activity with `quiet` once nothing +# is arriving, and fm-crew-state.sh only then withholds the recency note. Phases +# A and B pin both halves on one window - recent activity defers, the same pane +# with the activity gone escalates on the unchanged schedule - so a deferral can +# never become a standing exemption for a hung run or for a record left behind +# after the daemon exited under it. +test_busy_turn_bound_defers_only_while_the_run_reports_recent_activity() { + local dir state fakebin out drain_out capture_file window key pane_hash sig pid back + dir=$(make_case busy-run-activity); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; drain_out="$dir/drain.out"; capture_file="$dir/pane.txt" + window="test:fm-validating" + printf 'Working... (4210.6s)' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/validating.meta" + record_pi_busy "$state" validating + printf 'working: validating\n' > "$state/validating.status" + sig=$(seen_sig "$state/validating.status"); printf '%s' "$sig" > "$state/.seen-validating_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "Working... (4210.6s)") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + touch -t 200001010000 "$state/validating.turn-ended" + prime_turnend_seen "$state/validating.turn-ended" + # The bound crossed long ago and the idle window opened 500s ago, so the very + # first poll lands straight on the at-threshold branch that reads the evidence. + back=$(( $(date +%s) - 500 )) + echo "$back" > "$state/.stale-since-$key" + set_mtime "$back" "$state/.stale-since-$key" + + # Phase A: this crew's own run reports recent activity. Deferred. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing) · run activity recent' \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ + FM_PAUSE_RESURFACE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "a busy pane whose own run reports recent activity was wedge-escalated: $(cat "$out")" + fi + [ ! -s "$out" ] || { reap "$pid"; fail "a recent-activity deferral printed a wake reason: $(cat "$out")"; } + [ ! -s "$state/.wake-queue" ] || { reap "$pid"; fail "a recent-activity deferral enqueued a wake"; } + [ ! -e "$state/.wedge-escalations-$key" ] || { reap "$pid"; fail "a recent-activity deferral advanced the wedge escalation counter"; } + [ "$(cat "$state/.stale-since-$key" 2>/dev/null || echo 0)" -gt "$back" ] \ + || { reap "$pid"; fail "a recent-activity deferral did not restart the idle timer, so the next window cannot re-prove the run"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "could not acknowledge the intentional phase-A watcher stop" + + # Phase B: same window, same busy pane, same attributed working run-step - but + # the run has gone quiet (a hung step, or a record left behind after the daemon + # exited), so fm-crew-state.sh no longer marks its activity recent. + echo "$back" > "$state/.stale-since-$key" + set_mtime "$back" "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_FAKE_CREW_STATE='state: working · source: run-step · validating (fixing)' \ + FM_STATE_OVERRIDE="$state" FM_BUSY_TURN_MAX_SECS=1 FM_STALE_ESCALATE_SECS=240 \ + FM_PAUSE_RESURFACE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || fail "a quiet run behind a busy pane did not wedge-escalate past the turn-age bound: $(cat "$out")" + grep -F "stale: $window" "$out" >/dev/null || fail "the quiet-run escalation did not print the stale wake" + grep -F "possible wedge" "$out" >/dev/null || fail "the quiet-run escalation did not flag a possible wedge" + [ "$(cat "$state/.wedge-escalations-$key" 2>/dev/null || true)" = 1 ] || fail "the quiet-run escalation was not counted" + [ ! -e "$state/.stale-since-$key" ] || fail "the idle timer was not cleared after a real escalation" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || fail "drain after the quiet-run escalation failed" + grep "$(printf '\tstale\t')" "$drain_out" | grep -F "$window" >/dev/null || fail "the quiet-run escalation was not queued" + pass "a busy worker's wedge escalation is deferred only while its own no-mistakes run reports recent activity" +} + # --- declared pause + busy pane: the busy-turn bound must honor the declaration # A single foreground call can keep a declared external wait semantically busy # past the completed-turn bound, bypassing the ordinary stale-pause path. @@ -4432,6 +4508,7 @@ test_busy_pane_stable_hash_escalates_past_turn_age_bound test_busy_pane_changing_hash_escalates_past_turn_age_bound test_busy_pane_turn_end_touch_resets_age test_busy_pane_repeated_escalation_reaches_demand_deep_inspection +test_busy_turn_bound_defers_only_while_the_run_reports_recent_activity test_busy_pane_default_turn_age_bound_is_3600s test_busy_declared_pause_is_rechecked_not_wedge_escalated test_afk_busy_declared_pause_hands_off_plain_stale