From 836c804018718d8cc6497181bedc09fb9e485df4 Mon Sep 17 00:00:00 2001 From: dnth Date: Sat, 5 Sep 2026 18:07:16 +0800 Subject: [PATCH 1/5] fix(supervision): bound stale wake re-firing Port upstream #3672 stale acknowledgement protections to the OMP branch and carry the missing parked-worker cadence into the bash watcher. Add regression coverage for stale acknowledgements, OMP task scoping, and parked-worker wake bounds. --- .omp/extensions/fm-branch-supervision-omp.ts | 25 ++- .omp/extensions/lib/fm-branch-dispatch.ts | 23 ++- bin/fm-branch-prompt.sh | 2 + bin/fm-guard.sh | 19 +- bin/fm-wake-drain.sh | 49 ++++- bin/fm-watch.sh | 97 ++++++++-- docs/architecture.md | 19 +- docs/configuration.md | 5 +- docs/omp-supervision-branch.md | 2 + docs/watcher-continuity.md | 1 + tests/fm-omp-branch-supervision.test.sh | 18 ++ tests/fm-wake-queue.test.sh | 45 +++++ tests/fm-watch-triage.test.sh | 194 +++++++++++++++++++ 13 files changed, 459 insertions(+), 40 deletions(-) diff --git a/.omp/extensions/fm-branch-supervision-omp.ts b/.omp/extensions/fm-branch-supervision-omp.ts index 7cf8599b9eb..dc81ed2027e 100644 --- a/.omp/extensions/fm-branch-supervision-omp.ts +++ b/.omp/extensions/fm-branch-supervision-omp.ts @@ -411,6 +411,9 @@ function collectMainDialog(sessionManager: ReadonlyEntries, collection: MirrorCo export default function (pi: ExtensionAPI) { let branch: AgentSession | null = null; let branchBroken = ""; + // During a signal or stale prompt, reports are restricted to the tasks named + // by that prompt's granted rows. Heartbeat reviews remain unscoped. + let wakeTaskScope: { rows: string[]; tasks: Set } | null = null; let mainStreaming = false; let shuttingDown = false; // Advanced once per cold-start arm (session_start). There is no live-handoff @@ -449,6 +452,13 @@ export default function (pi: ExtensionAPI) { let mainModel: { provider: string; id: string } | null = null; let mainModelRegistry: ModelRegistry | null = null; + function wakeScopeRefusal(task: string): string { + if (!wakeTaskScope || wakeTaskScope.tasks.has(task)) return ""; + const named = [...wakeTaskScope.tasks].sort().join(", "); + const rows = wakeTaskScope.rows.join(", "); + return `report refused: the wake being handled (row ${rows}) names ${named}, not ${task}; report only that task, never fleet or a task from memory`; + } + // Main's own current effort needs no such tracking: OMP answers it directly // on demand, including at wake time. A value that is not one of the branch's // pinnable levels ("inherit", the auto sentinel, or undefined) means "follow @@ -688,6 +698,10 @@ export default function (pi: ExtensionAPI) { }; } const verdict = verdictRaw as Verdict; + const scopeRefusal = wakeScopeRefusal(task); + if (scopeRefusal) { + return { content: [{ type: "text", text: scopeRefusal }], details: undefined, isError: true }; + } const appendArgs = ["append", "--task", task, "--verdict", verdict, "--summary", summary, "--silent", String(silent)]; if (wake) appendArgs.push("--wake", wake); return enqueueDelivery(async () => { @@ -913,9 +927,14 @@ ${context.command} if (grant !== "published") throw new Error("could not record the branch's eligible row snapshot"); // A row can still arrive between this re-check and the model starting // the drain; that residual is accepted by the confused-agent-grade boundary. - await session.prompt( - `FIRSTMATE SUPERVISION WAKE: ${message}\n\nHandle this per your operating procedure. Do not finish this turn until you have completed, in order: fm_branch_report, the exact WAKE_ACK_REQUIRED command, and release of every task lease you claimed.`, - ); + wakeTaskScope = heartbeat ? null : { rows: [...scope.eligibleSeqs], tasks: new Set(scope.eligibleTasks) }; + try { + await session.prompt( + `FIRSTMATE SUPERVISION WAKE: ${message}\n\nHandle this per your operating procedure. Do not finish this turn until you have completed, in order: fm_branch_report, the exact WAKE_ACK_REQUIRED command, and release of every task lease you claimed.`, + ); + } finally { + wakeTaskScope = null; + } if (!(await releaseEligibleRowsSnapshot(state, wakeGrantScript, String(acceptedGeneration)))) { throw new Error("could not release the branch's settled wake-row grant"); } diff --git a/.omp/extensions/lib/fm-branch-dispatch.ts b/.omp/extensions/lib/fm-branch-dispatch.ts index 4aa42829602..cb8ecb36e57 100644 --- a/.omp/extensions/lib/fm-branch-dispatch.ts +++ b/.omp/extensions/lib/fm-branch-dispatch.ts @@ -33,6 +33,8 @@ export interface UnreadWakeScope { * `eligible` is false. */ eligibleSeqs: string[]; + /** Exact task ids named by the signal or stale rows granted to the branch. */ + eligibleTasks: string[]; /** * True only when this scan itself is untrustworthy: the queue or its * metadata could not be read, a line fails the structural tab-field check, @@ -47,8 +49,8 @@ export interface UnreadWakeScope { corrupted: boolean; } -const EMPTY_SCOPE: UnreadWakeScope = { status: "empty", eligible: false, projects: [], eligibleSeqs: [], corrupted: false }; -const UNSAFE_SCOPE: UnreadWakeScope = { status: "unsafe", eligible: false, projects: [], eligibleSeqs: [], corrupted: true }; +const EMPTY_SCOPE: UnreadWakeScope = { status: "empty", eligible: false, projects: [], eligibleSeqs: [], eligibleTasks: [], corrupted: false }; +const UNSAFE_SCOPE: UnreadWakeScope = { status: "unsafe", eligible: false, projects: [], eligibleSeqs: [], eligibleTasks: [], corrupted: true }; // scopeForUnreadWake is the single owner of branch-eligibility classification // (docs/omp-supervision-branch.md "Components and their owners" and @@ -92,6 +94,7 @@ export function scopeForUnreadWake(state: string, heartbeat: boolean): UnreadWak const projects = new Set(); const metadata = new Map(); + const taskByKey = new Map(); try { for (const name of readdirSync(state)) { if (!name.endsWith(".meta")) continue; @@ -101,7 +104,11 @@ export function scopeForUnreadWake(state: string, heartbeat: boolean): UnreadWak const window = fields.find((line) => line.startsWith("window="))?.slice(7) ?? ""; if (project) { metadata.set(task, project); - if (window) metadata.set(window, project); + taskByKey.set(task, task); + if (window) { + metadata.set(window, project); + taskByKey.set(window, task); + } } } } catch { @@ -109,6 +116,7 @@ export function scopeForUnreadWake(state: string, heartbeat: boolean): UnreadWak } const eligibleSeqs: string[] = []; + const eligibleTasks = new Set(); for (const line of rows) { const fields = line.split("\t"); if (fields.length < 5 || !/^[0-9]+$/.test(fields[1])) return UNSAFE_SCOPE; @@ -126,18 +134,21 @@ export function scopeForUnreadWake(state: string, heartbeat: boolean): UnreadWak continue; } let project = ""; + let task = ""; if (kind === "signal") { - const task = key.replace(/\.(?:status|turn-ended)$/, ""); + task = key.replace(/\.(?:status|turn-ended)$/, ""); project = metadata.get(task) ?? ""; } else if (kind === "stale") { + task = taskByKey.get(key) ?? taskByKey.get(key.replace(/^fm-/, "")) ?? ""; project = metadata.get(key) ?? metadata.get(key.replace(/^fm-/, "")) ?? ""; } else { // A kind fm_wake_append never emits: structural corruption, not an // ordinary main-only row. return UNSAFE_SCOPE; } - if (!project) return UNSAFE_SCOPE; + if (!project || !task) return UNSAFE_SCOPE; projects.add(project); + eligibleTasks.add(task); eligibleSeqs.push(seq); } const eligible = eligibleSeqs.length > 0; @@ -148,7 +159,7 @@ export function scopeForUnreadWake(state: string, heartbeat: boolean): UnreadWak // empty eligible set, so reading eligibility off the claim set rather than // off the heartbeat flag changes no pre-existing outcome and keeps a // heartbeat from being offered with nothing to hand over.) - return { status: eligible ? "safe" : "unsafe", eligible, projects: [...projects], eligibleSeqs, corrupted: false }; + return { status: eligible ? "safe" : "unsafe", eligible, projects: [...projects], eligibleSeqs, eligibleTasks: [...eligibleTasks], corrupted: false }; } // The exact state-relative filename bin/fm-wake-drain.sh reads for a diff --git a/bin/fm-branch-prompt.sh b/bin/fm-branch-prompt.sh index 4c0baf08b58..43f518b5ab7 100755 --- a/bin/fm-branch-prompt.sh +++ b/bin/fm-branch-prompt.sh @@ -92,6 +92,8 @@ Stay terse: your context is a cost. Do not re-read files the drain just printed. Never use shell background operators for supervision; the watcher and extension own continuity. Never call fm_branch_report speculatively - only after the event is actually handled or a refusal/lease conflict genuinely ended your handling. +The tool refuses a task the wake being handled did not name, fleet included (a heartbeat review is not scoped by task); a refusal means you reached for a task from memory, so report the wake's own task, never retry with another id. +An acknowledgement that consumed nothing says so and names the exact command for the current wake; run that printed command, do not drain again. # Recovery playbook (verbatim copy of the tracked skill) diff --git a/bin/fm-guard.sh b/bin/fm-guard.sh index cd5d5587421..9870a959c6b 100755 --- a/bin/fm-guard.sh +++ b/bin/fm-guard.sh @@ -25,7 +25,10 @@ # bounded). Independent alarms (queued wakes, worktree tangle) are never # suppressed by that dedup. Normal wake handling (watcher briefly down between a # wake and the next supervision resume) stays inside the grace window and stays -# silent. Always exits 0: the guard warns, it never blocks. +# silent. The queued-wakes warning stays silent for the supervision branch +# actor (FM_SUPERVISION_ACTOR=branch), because that actor runs guarded commands +# while handling exactly the queued rows its grant covers and can drain nothing +# else. Always exits 0: the guard warns, it never blocks. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -50,6 +53,12 @@ STALE_BANNER_MARKER="$STATE/.guard-watcher-stale-banner" . "$SCRIPT_DIR/fm-tangle-lib.sh" # shellcheck source=bin/fm-supervision-lib.sh . "$SCRIPT_DIR/fm-supervision-lib.sh" +# shellcheck source=bin/fm-lease-lib.sh +. "$SCRIPT_DIR/fm-lease-lib.sh" + +# The current actor (fm_lease_actor is the one owner of that identity); a +# malformed value is a wiring bug elsewhere, so the guard just warns as main. +GUARD_ACTOR=$(fm_lease_actor 2>/dev/null) || GUARD_ACTOR=main # Deterministic episode key from the qualitative down-state (the failing # condition), NOT the beacon mtime: under the auto-arm model a healthy @@ -230,10 +239,16 @@ fi # Queued wakes are an independent hazard; warn whenever they are pending, even if # a watcher is alive. Kept after the banner so the no-watcher alarm reads first. # Dedup of the watcher-down banner never suppresses this warning. +# The supervision branch is the exception: it runs guarded commands (fm-peek, +# fm-crew-state) in the middle of handling the very rows that are queued, and +# "drain them before anything else" mid-handling reads as "an earlier wake is +# still pending", which is what made it re-run a previous acknowledgement in a +# loop. The branch can act on nothing outside its grant anyway, so for that +# actor the guard stays silent about queued rows. if "$queue_pending"; then if [ "$READ_ONLY" -eq 1 ]; then echo "WARNING: queued wakes pending - left untouched because this session lacks verified fleet-lock ownership." >&2 - else + elif [ "$GUARD_ACTOR" != branch ]; then echo "WARNING: queued wakes pending - drain them with bin/fm-wake-drain.sh before anything else." >&2 fi fi diff --git a/bin/fm-wake-drain.sh b/bin/fm-wake-drain.sh index 2f49b1567cc..1b1e7a43ef5 100755 --- a/bin/fm-wake-drain.sh +++ b/bin/fm-wake-drain.sh @@ -32,6 +32,8 @@ RECOVERY_ACK_REQUIRED=false RECOVERY_ACK_MOVED=false ACK_THROUGH= ACK_GENERATION= +ACK_REMOVED=0 +PRESENTED_MAX=0 # --- per-actor consume (docs/omp-supervision-branch.md "Per-actor acknowledgement") -- # main (FM_SUPERVISION_ACTOR unset or "main", via fm-lease-lib.sh's fm_lease_actor @@ -155,6 +157,19 @@ decide_scoped_locked() { fi } +# The highest sequence this actor has already been presented: the branch's +# grant is exactly its current prompt's rows, and main's claim file is what its +# last drain printed. Read BEFORE an ack re-claims, so a row that arrived since +# presentation is never named as "the current wake" the caller may acknowledge +# unseen. 0 when nothing is on record. +presented_max_row() { # + if rows_file_valid "$1" 2>/dev/null; then + awk '$1 ~ /^[0-9]+$/ && $1 > max { max=$1 } END { print max + 0 }' "$1" + else + printf '0\n' + fi +} + case "${1:-}" in '') ;; --ack-through) @@ -329,6 +344,13 @@ if [ "$SCOPED" = true ]; then fi if [ -n "$ACK_THROUGH" ]; then + if [ "$SCOPED" = true ]; then + if [ "$ACTOR" = branch ]; then + PRESENTED_MAX=$(presented_max_row "$ELIGIBLE_ROWS_FILE") || exit 1 + else + PRESENTED_MAX=$(presented_max_row "$MAIN_ROWS_FILE") || exit 1 + fi + fi # Row consumption is bound to the monotonic sequence and always happens. # Only retiring the episode is bound to the generation, so a generation that # moved on names its own remedy instead of refusing and consuming nothing. @@ -361,7 +383,10 @@ if [ -n "$ACK_THROUGH" ]; then NF < 5 || $2 !~ /^[0-9]+$/ || $2 > cutoff || !($2 in owned) { print } ' "$FM_WAKE_QUEUE" > "$DRAIN_TMP" || exit 1 fi + ACK_REMOVED=$(( $(awk 'END { print NR }' "$FM_WAKE_QUEUE") - $(awk 'END { print NR }' "$DRAIN_TMP") )) if [ ! -s "$DRAIN_TMP" ]; then + fm_recovery_marker_snapshot "$RECOVERY_MARKER" || exit 1 + RECOVERY_MARKER_TOKEN=$FM_RECOVERY_MARKER_TOKEN fm_recovery_marker_ack "$RECOVERY_MARKER" "$ACK_GENERATION" RECOVERY_ACK_STATUS=$? case "$RECOVERY_ACK_STATUS" in @@ -393,9 +418,27 @@ if [ -n "$ACK_THROUGH" ]; then fi fm_lock_release "$FM_WAKE_QUEUE_LOCK" DRAIN_LOCK_HELD=false - if [ "$RECOVERY_ACK_MOVED" = true ]; then - printf 'wake drain: acknowledged wakes through %s, but a newer recovery episode is pending; re-run bin/fm-wake-drain.sh and use the new WAKE_ACK_REQUIRED command\n' \ - "$ACK_THROUGH" >&2 + if [ "$ACK_REMOVED" -eq 0 ] && [ "$PRESENTED_MAX" -gt "$ACK_THROUGH" ]; then + # Nothing at or below the cutoff was this actor's to consume, while a + # presented row above it is still waiting: the caller acknowledged an + # earlier wake, not the one it is handling. Say so, and name the exact + # command for the current wake, so the remedy is never "drain again" (which + # re-presents the same row and invites the same stale acknowledgement). + # The generation is the marker's current one; only a retired marker cannot + # be named because the next drain opens a fresh generation for it. + case "$RECOVERY_MARKER_TOKEN" in + pending:*|announced:*) + printf 'wake drain: nothing was acknowledged through %s (none of your presented wake rows is at or below it); the current wake is row %s: run bin/fm-wake-drain.sh --ack-through %s --recovery-generation %s after handling it\n' \ + "$ACK_THROUGH" "$PRESENTED_MAX" "$PRESENTED_MAX" "${RECOVERY_MARKER_TOKEN##*:}" >&2 + ;; + *) + printf 'wake drain: nothing was acknowledged through %s (none of your presented wake rows is at or below it); the current wake is row %s: re-run bin/fm-wake-drain.sh and use the WAKE_ACK_REQUIRED command it prints\n' \ + "$ACK_THROUGH" "$PRESENTED_MAX" >&2 + ;; + esac + elif [ "$RECOVERY_ACK_MOVED" = true ]; then + printf 'wake drain: acknowledged wakes through %s (%s row(s) consumed), but a newer recovery episode is pending; re-run bin/fm-wake-drain.sh and use the new WAKE_ACK_REQUIRED command\n' \ + "$ACK_THROUGH" "$ACK_REMOVED" >&2 fi exit 0 fi diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index f1dd6bb1423..3b3069572fe 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -216,7 +216,10 @@ BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # A crew that declared a pause is idling on a known external wait, so its stale # pane is absorbed rather than wedge-escalated. # A captain-held or paused crew whose agent has confidently exited uses the same -# bounded cadence, while a live or ambiguously read agent still surfaces once. +# bounded cadence, while a live or ambiguously read agent surfaces on first sight +# and is then held to that same cadence; a secondmate earns the cadence on its +# declaration alone, because its endpoint liveness is deliberately never read +# (pause_state_class owns that split). # These cases re-surface once for a recheck every PAUSE_RESURFACE_SECS - far # longer than the wedge threshold, but finite so a forgotten hold cannot rot invisibly. PAUSE_RESURFACE_SECS=${FM_PAUSE_RESURFACE_SECS:-$FM_PAUSE_RESURFACE_SECS_DEFAULT} @@ -402,16 +405,21 @@ FM_WEDGE_DEMAND_INSPECT_COUNT=${FM_WEDGE_DEMAND_INSPECT_COUNT:-3} # absorb can rot invisibly. is how long the current absorb has held and # is the per-window marker whose mtime records the last re-surface, so # once past PAUSE_RESURFACE_SECS the pane wakes once per window rather than every -# poll. Shared by the declared-pause absorb and the worktree-write deferral so the -# two cadences cannot drift apart; each caller owns its own marker and reason. +# poll. An optional binds that cadence to its current declaration; callers +# without a scoped declaration keep the timestamp body. Shared by the +# declared-pause absorb and the worktree-write deferral so the two cadences cannot +# drift apart; each caller owns its own marker and reason. # Returns without waking while either the absorb or the throttle is inside the # window; wake() itself exits the cycle, exactly as it does inline. -resurface_absorbed() { # - local win=$1 throttle=$2 age=$3 reason=$4 - [ "$age" -ge "$PAUSE_RESURFACE_SECS" ] || return 0 - [ "$(age_of "$throttle")" -ge "$PAUSE_RESURFACE_SECS" ] || return 0 # 999999 when no prior re-surface +resurface_absorbed() { # [scope] + local win=$1 throttle=$2 age=$3 reason=$4 scope=${5-} + if [ -z "$scope" ] || [ ! -e "$throttle" ] \ + || [ "$(cat "$throttle" 2>/dev/null || true)" = "$scope" ]; then + [ "$age" -ge "$PAUSE_RESURFACE_SECS" ] || return 0 + [ "$(age_of "$throttle")" -ge "$PAUSE_RESURFACE_SECS" ] || return 0 # 999999 when no prior re-surface + fi fm_wake_append stale "$win" "$reason" || exit 1 - date +%s > "$throttle" + if [ -n "$scope" ]; then printf '%s' "$scope" > "$throttle"; else date +%s > "$throttle"; fi wake "$reason" } @@ -515,7 +523,7 @@ busy_turn_over_age() { # # above, throttled by this window's own .paused-resurfaced- marker. Advances # the stale suppressor to and flags the key paused. handle_paused_stale() { # - local win=$1 task=$2 h=$3 key statusf mtime age + local win=$1 task=$2 h=$3 key statusf mtime age declaration key=$(window_key "$win") printf '%s' "$h" > "$STATE/.stale-$key" : > "$STATE/.paused-$key" @@ -525,8 +533,9 @@ handle_paused_stale() { # mtime=$(stat_mtime "$statusf") case "$mtime" in ''|*[!0-9]*) mtime=$(date +%s) ;; esac age=$(( $(date +%s) - mtime )) - resurface_absorbed "$win" "$STATE/.paused-resurfaced-$key" "$age" \ - "stale: $win (paused ${age}s, awaiting external - declared pause, rechecked on a long cadence not a wedge; confirm the wait still holds)" + reason="paused ${age}s, awaiting external - declared pause, rechecked on a long cadence not a wedge; confirm the wait still holds" + declaration="declared:$(stat_sig "$statusf" || true)" + resurface_absorbed "$win" "$STATE/.paused-resurfaced-$key" "$age" "stale: $win ($reason)" "$declaration" triage_log "absorbed stale (paused, awaiting external, age ${age}s): $win" } @@ -588,13 +597,22 @@ clear_pause_state() { # rm -f "$STATE/.paused-$key" "$STATE/.paused-rechecked-$key" "$STATE/.paused-resurfaced-$key" } -clear_pause_tracking() { # +# The hash-scoped half of clear_pause_tracking: the stale suppressor, its wedge +# timer and escalation count, and the write-deferral chain. Split out so a caller +# that must keep a window's DECLARATION-scoped pause state - its .paused-* flag, +# recheck, and re-surface throttle - can still reset the per-hash half alone. +clear_stale_hash_tracking() { # local key=$1 - clear_pause_state "$key" clear_write_tracking "$key" rm -f "$STATE/.stale-$key" "$STATE/.stale-since-$key" "$STATE/.wedge-escalations-$key" } +clear_pause_tracking() { # + local key=$1 + clear_pause_state "$key" + clear_stale_hash_tracking "$key" +} + # Reconcile a declared pause or captain-held status with authoritative crew state. # Only a confidently dead ordinary crew may recover paused classification after # fm-crew-state has fallen back to stopped or unknown. @@ -642,21 +660,51 @@ pause_state_class() { # printf '%s' "$class" } +# Surface a stale pane no classifier could resolve, so firstmate inspects it: it +# may have finished through an interactive menu that wrote no status, be waiting on +# a decision, or be wedged. pause_state_class deliberately answers `none` for a +# still-LIVE agent even under a declared wait, so a worker genuinely waiting on a +# decision is never silenced - which routes every parked-but-live worker here, on +# first sight of each distinct stale hash. +# +# So a declared wait bounds this path to the same once-per-PAUSE_RESURFACE_SECS +# cadence resurface_absorbed owns for the absorbed paths, throttled by this +# window's own .paused-resurfaced- marker: an idle parked pane still churns +# its hash (a clock, a token counter), and each new hash re-enters this path, so +# without that bound one declared wait re-alarms firstmate for its whole duration. +# The FIRST sight still wakes, keeping the inspect-an-inconclusive-state intent, +# and the throttle is read BEFORE anything is queued and advanced only by a wake +# that really fires - a throttle written by the wake it should have prevented, or +# read after that wake was already appended, bounds nothing. surface_nonterminal_stale() { # - local win=$1 h=$2 key task last + local win=$1 h=$2 key task last declaration='' declared=1 throttled=1 key=$(window_key "$win") - fm_wake_append stale "$win" "stale: $win" || exit 1 - printf '%s' "$h" > "$STATE/.stale-$key" - rm -f "$STATE/.stale-since-$key" - clear_write_tracking "$key" task=$(window_to_task "$win" "$STATE") last=$(last_status_line "$STATE/$task.status") if status_is_paused_or_captain_held "$last"; then + declared=0 + declaration="declared:$(stat_sig "$STATE/$task.status" || true)" + if [ "$(cat "$STATE/.paused-resurfaced-$key" 2>/dev/null || true)" = "$declaration" ] \ + && [ "$(age_of "$STATE/.paused-resurfaced-$key")" -lt "$PAUSE_RESURFACE_SECS" ]; then + throttled=0 + fi + fi + if [ "$throttled" -ne 0 ]; then + fm_wake_append stale "$win" "stale: $win" || exit 1 + fi + printf '%s' "$h" > "$STATE/.stale-$key" + rm -f "$STATE/.stale-since-$key" + clear_write_tracking "$key" + if [ "$declared" -eq 0 ]; then : > "$STATE/.paused-$key" date +%s > "$STATE/.paused-rechecked-$key" - date +%s > "$STATE/.paused-resurfaced-$key" + [ "$throttled" -eq 0 ] || printf '%s' "$declaration" > "$STATE/.paused-resurfaced-$key" else - rm -f "$STATE/.paused-$key" "$STATE/.paused-rechecked-$key" "$STATE/.paused-resurfaced-$key" + clear_pause_state "$key" + fi + if [ "$throttled" -eq 0 ]; then + triage_log "absorbed non-terminal stale (declared wait already re-surfaced this window): $win" + return 0 fi wake "stale: $win" } @@ -1505,6 +1553,15 @@ EOF if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; + # Inconclusive, but the declared wait itself still stands, so only the + # per-hash bookkeeping resets. The re-surface throttle bounds the + # DECLARATION, not the pane hash: an idle parked pane whose display + # ticks (a clock, a token counter) changes hash without changing what + # is being waited on, and clearing the throttle here would hand that + # same wait a fresh window on every tick - the first sight of each new + # hash reaches surface_nonterminal_stale below, so the whole declared + # wait would re-alarm far inside PAUSE_RESURFACE_SECS. + none) clear_stale_hash_tracking "$key" ;; *) clear_pause_tracking "$key" ;; esac elif [ "$paused_bound" -ne 0 ] && [ -e "$pf" ]; then diff --git a/docs/architecture.md b/docs/architecture.md index 09548a67eba..a95c74339e8 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -26,10 +26,21 @@ The receipt makes retirement safely retryable across restarts: fixed-path recove A concurrent replacement remains armed, every non-merged or invalid observation remains unchanged, and retirement never performs task or persistent-secondmate cleanup. `bin/fm-pr-lib.sh` owns the notification-marker and retirement-receipt formats plus their strict identity mechanics, [`bin/fm-merge-outcome-lib.sh`](../bin/fm-merge-outcome-lib.sh) owns role-routed publication, the local durable row, and marker ordering, and `bin/fm-watch.sh` owns immediate poll-result delivery and retirement. A merge-outcome row is an ordinary check-kind wake, so the OMP supervision branch's dispatch classifier keeps it main-owned exactly like every other merge-confirmation poll result. -No-verb wakes, such as `working:` notes and bare turn-ended signals, are benign only when `bin/fm-crew-state.sh` reports positive evidence that the crew is still working: a currently attributed active no-mistakes step, or an exact busy verdict from the semantic busy-state contract. -A crew that declares `paused:` for a known external wait is separately absorbed while idle and re-surfaced only on the longer pause cadence, rather than being treated as a possible wedge. -For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint only when the backend confidently reports its agent dead. -Live or inconclusive liveness remains fail-open at that initial surface. +No-verb wakes, such as `working:` notes and bare turn-ended signals, are benign only when every referenced task independently has positive evidence that its crew is still working: a currently attributed active no-mistakes step, or an exact busy verdict from the semantic busy-state contract, both read through `bin/fm-crew-state.sh`. +A home that creates `config/turnend-churn-absorb` lets each eligible bare turn-ended task that lacks either authoritative proof use a third form: pane content that changed since the previous poll, compared against the same `state/.hash-*` marker the staleness backbone records, which claims no harness semantics and needs no adapter cooperation. +That form stays opt-in because it infers execution from rendered bytes rather than from a verdict the harness vouches for, so with the flag absent triage behaves exactly as it did before ([`configuration.md`](configuration.md) "Turn-end pane-churn absorb"). +That evidence clears the pane's prior stale classification and wedge-escalation count, then defers such a wake rather than swallowing it, since a crew that has stopped renders nothing further and its now-static pane surfaces through the staleness backbone within a poll or two, even if its final bytes match an earlier stale render. +A wake naming any status file remains governed solely by the strict authoritative proof, and the pane-churn fallback is unavailable to an entire batch that references a secondmate. +An unresolvable endpoint, an ambiguous marker key, a missing or malformed prior hash, a capture that fails or returns empty, an invalid deferral bound or deadline, or an unwritable deferral marker surfaces without clearing prior stale classification. +The deferral is bounded per endpoint by `FM_TURNEND_CHURN_ABSORB_SECS`, tracked in `state/.churn-since-*`, after which the turn-end surfaces and the window restarts. +That bound is load-bearing rather than cosmetic: churn and staleness read the same pane, so a pane that renders continuously - a clock, a spinner, a shell heartbeat, or a harness that leaves a background renderer alive after its agent yields - never reaches the staleness backbone's two-identical-hashes test either, and an unbounded churn absorb would leave a genuinely stopped worker behind such a renderer with no path left to surface it. +If two metadata records derive the same per-window marker key, including two records that name the same endpoint, that marker is not attributable churn evidence for either task, so the bare turn-ended wake surfaces without changing or migrating existing marker state. +A `kind=secondmate` task's status signal is the parent-directed reply stream and is never absorbed as provably working; its bare turn-ended signal is absorbed only by the ordinary authoritative working proof because an active secondmate does not enter the staleness backbone that would resurface deferred pane-churn evidence. +A crew that declares `paused:` for a known external wait, or carries a verified `captain-held` transfer, is separately absorbed while idle and re-surfaced only on the longer pause cadence, rather than being treated as a possible wedge. +For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint; the pause classification itself is recovered only when the backend confidently reports its agent dead. +Live or inconclusive liveness remains fail-open at that initial surface, so a worker genuinely waiting on a decision is never silenced. +Its later sights are still held to that same bounded cadence rather than re-alarming on every pane-hash change, because the throttle is keyed to the declaration and not to the pane an idle parked worker keeps ticking. +A secondmate's endpoint liveness is still never read at all; a mate is admitted to that same cadence only to serve a declared wait's bounded re-surface, so a forgotten pause or captain hold on a mate cannot rot invisibly. The emitted [OMP supervision protocol](supervision-protocols/omp.md) owns the separate bounded home-watcher-beacon rule for quiet secondmate endpoints. Its initial normal-mode status signal still surfaces through the no-verb path, while away mode self-handles that routine signal and owns the later recheck. Fresh stale panes use the same current-state read before trusting the status log, so an active run or a proven busy worker outranks an old captain-relevant status-log line left behind before validation. diff --git a/docs/configuration.md b/docs/configuration.md index a6bf0cd8050..2e04f48139a 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -700,10 +700,11 @@ FM_WATCHER_STALE_GRACE=300 # defaults to FM_GUARD_GRACE; seconds a live watche FM_SIGNAL_GRACE=30 # seconds to coalesce nearby status and turn-end signals into one wake FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked -FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless they declare pause or verified dead/missing captain-held recovery +FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless admitted directly to the declared-wait cadence, while a live idle declared wait still surfaces once before that cadence bounds repeats FM_REMOTE_STALE_RECHECK_SECS=60 # seconds between inconclusive remote stale-owner probes during away-mode recovery; invalid values use 60 FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's live-generation state/.turn-ended. marker, or its state/.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait or verified captain-held transfer takes the FM_PAUSE_RESURFACE_SECS recheck below instead -FM_PAUSE_RESURFACE_SECS=2700 # seconds before the watcher re-surfaces a declared external wait or verified captain-held transfer for a recheck, including a live busy pane past FM_BUSY_TURN_MAX_SECS; the away-mode daemon uses the same setting for declared external waits and verified local captain-held recovery +FM_PAUSE_RESURFACE_SECS=2700 # seconds between bounded rechecks of a declared external wait or verified captain-held transfer, including a live idle pane after its first inconclusive stale wake and a live busy pane past FM_BUSY_TURN_MAX_SECS; the away-mode daemon uses the same setting, ageing its window against the crew's own latest status line rather than pane busy state +FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless admitted directly to the declared-wait cadence, while a live idle declared wait still surfaces once before that cadence bounds repeats FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added FM_WORKTREE_WRITE_PRUNE='.git node_modules .venv venv __pycache__ .mypy_cache .pytest_cache .ruff_cache .tox target dist build .next .cache vendor' # directory names the wedge detector's task-worktree write probe skips; the default keeps .git out so a supervisor's own read-only git command can never look like crew progress; set it to the empty string to prune nothing, which widens the probe to the whole depth-bounded tree rather than disabling it FM_WORKTREE_WRITE_MAXDEPTH=6 # depth that same probe walks below the recorded worktree; it runs only at the moment a wedge escalation would otherwise fire, never on every poll; no probe knob applies to a secondmate, whose recorded worktree is a provisioned home the probe skips entirely diff --git a/docs/omp-supervision-branch.md b/docs/omp-supervision-branch.md index dce6f1297f4..64308d9afe5 100644 --- a/docs/omp-supervision-branch.md +++ b/docs/omp-supervision-branch.md @@ -24,6 +24,7 @@ This feature is OMP-only by construction and changes nothing anywhere else: It checks the current extension generation and `state/.lock` ownership before each guarded branch side effect, so a lost lock or a cold-start re-arm cannot let an old continuation mutate the fleet. The branch conversation remains resident across ordinary main turns. When main performs `/new`, `/resume`, `/fork`, or reload, OMP emits `session_switch`; the primary watcher retires the prior generation and re-arms its replacement before the next model turn. The mirror re-anchors when main's session file changes, and any actionable close whose delivery overlaps the replacement is carried to the new generation exactly once. `session_shutdown` remains the terminal-process boundary rather than a replacement signal. Every accepted path that cannot reach a working branch rejects its settlement to the shared watcher core, which retains delivery ownership until main consumes the follow-up; a broken branch declines later offers so they take that same watcher-owned path directly. + While a signal or stale prompt is open, `fm_branch_report` accepts only the task ids resolved from that prompt's granted rows, and refuses `fleet` or remembered task ids; heartbeat reviews remain unscoped. - Branch model and effort selection: the same extension registers `/supervision-model`, which picks the branch's model and then its reasoning effort over OMP's portable select dialog, and saves both as pins applied at the next branch build, never to the branch already running; [configuration.md](configuration.md#omp-supervision-branch-model-and-effort-configsupervision-branch-model-configsupervision-branch-effort) owns the full activation boundary, operator-facing schema, and behavior. Model resolution reads OMP's live `ModelRegistry` (available models and their configured credentials) with pure lookups, so pinning the branch never moves main's own conversation; effort uses OMP's `Effort` catalog and the shim's `clampThinkingLevel`. - Branch system prompt: `bin/fm-branch-prompt.sh`; its header owns the byte-stable-prefix contract (no timestamps, no fleet snapshot, no per-wake content). @@ -60,6 +61,7 @@ Main claims every unread row not currently granted to the branch, then drains an The branch drains and acknowledges only the exact row set the extension granted to it, published to `state/.branch-eligible-rows` under the queue lock immediately before every branch prompt. `.omp/extensions/lib/fm-branch-dispatch.ts`'s `scopeForUnreadWake` is the single owner of which rows are branch-eligible; the drain never reclassifies a row itself, it only consumes that already-computed snapshot. A row whose sequence number is not in the branch's snapshot is left completely untouched by a branch-actor drain or acknowledgement, no matter its sequence number, so the branch can never swallow a main-owned row still waiting for main. +An acknowledgement that consumes none of the actor's presented rows reports that fact and names the exact current `--ack-through` and `--recovery-generation` command, so retrying an earlier wake cannot re-fire a stale loop. This scoping engages only while a branch grant is, or recently was, in play: a home that never runs the branch has none of the actor files, so the drain takes its byte-behavior-identical pre-branch path and writes no actor state. ## Main-fallback re-entry limitation diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 9334422ab13..7a96341424b 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -62,6 +62,7 @@ Every watcher close and every durable queue append publishes downtime, so a down That reuse keeps a watcher close inside the handling window from orphaning the acknowledgement already presented and trapping later arms in repeated recovery presentation. An acknowledgement carries two separable facts: queue-row consumption is bound to the monotonic `--ack-through` sequence, while only retiring the episode is bound to `--recovery-generation`. A generation mismatch therefore does not block consumption of rows through that sequence; it is a non-fatal result that names its own remedy - re-drain, then acknowledge the newer episode. +When that cutoff consumes none of the actor's presented rows while a newer presented row remains, the drain instead says that nothing was acknowledged and prints the exact command for the current wake, preventing a stale acknowledgement from re-feeding the same loop. The acknowledgement retires the marker only when no rows remain after sequence-bound consumption. A concurrently appended wake has a higher sequence, remains queued, and keeps the episode pending for presentation. Consequently, an empty-queue downtime publication during handling can be retired by the outstanding acknowledgement without a dedicated recovery turn. diff --git a/tests/fm-omp-branch-supervision.test.sh b/tests/fm-omp-branch-supervision.test.sh index 77ad152915f..902013b9b13 100755 --- a/tests/fm-omp-branch-supervision.test.sh +++ b/tests/fm-omp-branch-supervision.test.sh @@ -406,6 +406,23 @@ JS pass "OMP outcome delivery yields to the event loop while preserving routine/captain order" } +test_omp_scope_resolves_reportable_tasks_from_granted_rows() { + local state out + state="$TMP_ROOT/scope-task-resolution/state" + mkdir -p "$state" + printf 'project=project-a\nwindow=default:wA:p1\n' > "$state/task-a.meta" + printf '1\t1\tsignal\ttask-a.status\tsignal: task-a\n2\t2\tstale\tdefault:wA:p1\tstale: default:wA:p1\n3\t3\tcheck\tmerge-poll\tcheck: merged\n' > "$state/.wake-queue" + out=$(STATE_PATH="$state" DISPATCH_PATH="$ROOT/.omp/extensions/lib/fm-branch-dispatch.ts" node --experimental-strip-types --input-type=module -e ' + const { scopeForUnreadWake } = await import(process.env.DISPATCH_PATH); + const scope = scopeForUnreadWake(process.env.STATE_PATH, false); + if (JSON.stringify(scope.eligibleSeqs) !== JSON.stringify(["1", "2"])) throw new Error(JSON.stringify(scope)); + if (JSON.stringify([...scope.eligibleTasks].sort()) !== JSON.stringify(["task-a"])) throw new Error(JSON.stringify(scope)); + console.log("scope resolved"); + ' 2>&1) || fail "OMP dispatch scope failed: $out" + assert_contains "$out" "scope resolved" "OMP dispatch scope did not resolve reportable tasks" + pass "OMP dispatch resolves signal and stale rows to the branch report task and excludes checks" +} + test_branch_prompt_is_byte_stable_and_above_cache_floor test_outcome_store_is_append_only_with_cursor_reads test_outcome_startup_replay_preserves_silence @@ -417,3 +434,4 @@ test_pr_check_lease_guard_serializes_task_metadata test_non_branch_home_is_untouched test_omp_extension_establishes_main_actor_context test_async_outcome_delivery_keeps_event_loop_responsive_and_ordered +test_omp_scope_resolves_reportable_tasks_from_granted_rows diff --git a/tests/fm-wake-queue.test.sh b/tests/fm-wake-queue.test.sh index 3406fc1733d..2fc58219df8 100755 --- a/tests/fm-wake-queue.test.sh +++ b/tests/fm-wake-queue.test.sh @@ -902,6 +902,50 @@ test_branch_owner_activation_rollback_stops_after_publication() { pass "branch activation rollback succeeds before publication and preserves every active grant" } +# A stale acknowledgement for an earlier presented wake must not re-feed the +# same wake loop: it consumes nothing, names the current wake, and prints the +# exact generation-bound command that closes that current presentation. +test_stale_acknowledgement_names_current_presented_wake() { + local dir state grant first_seq first_gen second_seq second_gen + dir=$(make_case stale-ack-current-wake) + state="$dir/state" + grant="$ROOT/bin/fm-wake-grant.sh" + append_wake "$state" signal task-a.status "signal: first wake" || fail "first wake append failed" + FM_STATE_OVERRIDE="$state" "$grant" activate "$$" stale-ack || fail "branch owner activation failed" + FM_STATE_OVERRIDE="$state" "$grant" publish stale-ack 1 || fail "first grant publication failed" + FM_STATE_OVERRIDE="$state" FM_SUPERVISION_ACTOR=branch "$DRAIN" > "$dir/first.out" 2> "$dir/first.err" || fail "first drain failed" + first_seq=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through \([0-9][0-9]*\) --recovery-generation .*/\1/p' "$dir/first.err") + first_gen=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--recovery-generation \([A-Za-z0-9._-][A-Za-z0-9._-]*\)$/\1/p' "$dir/first.err") + [ -n "$first_seq" ] && [ -n "$first_gen" ] || fail "first drain omitted its acknowledgement command" + FM_STATE_OVERRIDE="$state" FM_SUPERVISION_ACTOR=branch "$DRAIN" --ack-through "$first_seq" --recovery-generation "$first_gen" || fail "first acknowledgement failed" + FM_STATE_OVERRIDE="$state" "$grant" release stale-ack || fail "first grant release failed" + + append_wake "$state" stale fm-window-b "stale: second wake" || fail "second wake append failed" + FM_STATE_OVERRIDE="$state" "$grant" publish stale-ack 2 || fail "second grant publication failed" + FM_STATE_OVERRIDE="$state" FM_SUPERVISION_ACTOR=branch "$DRAIN" > "$dir/second.out" 2> "$dir/second.err" || fail "second drain failed" + second_seq=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through \([0-9][0-9]*\) --recovery-generation .*/\1/p' "$dir/second.err") + second_gen=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through [0-9][0-9]* --recovery-generation \([A-Za-z0-9._-][A-Za-z0-9._-]*\)$/\1/p' "$dir/second.err") + [ "$second_seq" -gt "$first_seq" ] || fail "second drain did not present a newer wake" + + FM_STATE_OVERRIDE="$state" FM_SUPERVISION_ACTOR=branch "$DRAIN" --ack-through "$first_seq" --recovery-generation "$first_gen" \ + > "$dir/stale.out" 2> "$dir/stale.err" || fail "stale acknowledgement failed unsafely" + grep -F "nothing was acknowledged through $first_seq" "$dir/stale.err" >/dev/null \ + || fail "stale acknowledgement did not report that it consumed nothing: $(cat "$dir/stale.err")" + grep -F "the current wake is row $second_seq" "$dir/stale.err" >/dev/null \ + || fail "stale acknowledgement did not name the current wake: $(cat "$dir/stale.err")" + ! grep -F 're-run bin/fm-wake-drain.sh' "$dir/stale.err" >/dev/null \ + || fail "stale acknowledgement invited the old drain loop: $(cat "$dir/stale.err")" + grep -F "run bin/fm-wake-drain.sh --ack-through $second_seq --recovery-generation $second_gen" "$dir/stale.err" >/dev/null \ + || fail "stale acknowledgement omitted the exact current command: $(cat "$dir/stale.err")" + grep -F "$(printf '\tstale\tfm-window-b\t')" "$state/.wake-queue" >/dev/null \ + || fail "stale acknowledgement consumed the current wake" + + FM_STATE_OVERRIDE="$state" FM_SUPERVISION_ACTOR=branch "$DRAIN" --ack-through "$second_seq" --recovery-generation "$second_gen" \ + || fail "the printed current-wake command failed" + [ ! -s "$state/.wake-queue" ] || fail "the current wake remained queued after its exact acknowledgement" + pass "a stale acknowledgement is bounded and names the current presented wake" +} + # Consumer-side incarnation gate for turn-end markers (fm-wake-lib.sh). # bin/fm-turnend-signal.sh writes state/.turn-ended lock-free and # unconditionally, stamped with the firing spawn_gen. The consumer discards a @@ -947,6 +991,7 @@ test_turnend_marker_consumer_incarnation_gate() { } test_turnend_marker_consumer_incarnation_gate +test_stale_acknowledgement_names_current_presented_wake test_concurrent_append_and_drain test_signal_catchup_without_running_watcher test_stale_enqueue_before_suppressor diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index 4f6c7f9114a..7a579ca511d 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -1209,6 +1209,198 @@ test_exited_declared_pause_is_bounded_but_live_gate_surfaces() { pass "exited declared-pause and captain-held panes use bounded pause cadence while a live decision gate still surfaces once" } +# A dead worker reaches handle_paused_stale rather than the live fallback above. +# When one declared wait directly replaces another, the existing +# throttle belongs to the old declaration and must not suppress the new wait's +# first inspection merely because its timestamp is still young. +test_absorbed_replacement_wait_does_not_inherit_the_old_throttle() { + local spec name initial replacement expected dir state fakebin out capture_file + local statusf window key sig back pid wakes + for spec in \ + 'paused-replacement|paused: waiting on validation run one|paused: waiting on validation run two|awaiting external' \ + 'captain-held-replacement|captain-held [key=route]: awaiting the routing call|captain-held [key=release]: awaiting the release call|awaiting external' + do + name=${spec%%|*}; spec=${spec#*|} + initial=${spec%%|*}; spec=${spec#*|} + replacement=${spec%%|*}; expected=${spec#*|} + dir=$(make_case "$name"); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; statusf="$state/held.status" + window="test:fm-held" + printf 'idle after agent exit\n' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=grok\nbackend=tmux\n' "$window" > "$state/held.meta" + printf '%s\n' "$initial" > "$statusf" + back=$(( $(date +%s) - 500 )) + if [ "$(uname)" = Darwin ]; then touch -mt "$(date -r "$back" '+%Y%m%d%H%M.%S')" "$statusf" + else touch -m -d "@$back" "$statusf"; fi + sig=$(seen_sig "$statusf"); printf '%s' "$sig" > "$state/.seen-held_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + printf '%s' "$(hash_text 'idle after agent exit')" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_FAKE_TMUX_CURRENT_COMMAND=zsh FM_FAKE_CREW_STATE='state: stopped · source: pane · bare shell' \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_PAUSE_RESURFACE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" >> "$out" & + pid=$! + wait_for_exit "$pid" 100 || fail "[$name] initial declared wait did not re-surface" + ack_stopped_cycle "$state" || fail "[$name] could not acknowledge the initial declared wait" + + printf '%s\n' "$replacement" >> "$statusf" + sig=$(seen_sig "$statusf"); printf '%s' "$sig" > "$state/.seen-held_status" + printf 'idle after replacement wait\n' > "$capture_file" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_FAKE_TMUX_CURRENT_COMMAND=zsh FM_FAKE_CREW_STATE='state: stopped · source: pane · bare shell' \ + FM_WATCH_HANDLING_SUCCESSOR=1 \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_PAUSE_RESURFACE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" >> "$out" & + pid=$! + wait_for_exit "$pid" 100 \ + || { reap "$pid"; fail "[$name] replacement declared wait inherited the old throttle"; } + wakes=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w { n++ } END { print n + 0 }' \ + "$state/.wake-queue" 2>/dev/null || echo 0) + [ "$wakes" -eq 1 ] || fail "[$name] replacement declared wait produced $wakes wakes instead of one" + grep -F "$expected" "$state/.wake-queue" >/dev/null \ + || fail "[$name] replacement declared wait used the wrong recheck reason: $(cat "$state/.wake-queue")" + done + pass "absorbed paused and captain-held replacements each start their own re-surface cadence" +} + +# Run one watcher round against a parked-worker fixture, so a round differs only +# in the pane contents the case just wrote. Armed the way fm-watch-arm.sh arms a +# successor after firstmate handled a wake, because that is what a supervision +# turn actually does and it is the only arm that stays in the poll loop instead of +# re-announcing the previous round's downtime - without it a round exits on +# `check: rearm-resurface` before it ever reaches the stale path, and every +# absorb assertion below passes vacuously. A live agent (pane_current_command +# matching the recorded harness) on an idle pane is the exact population +# pause_state_class answers `none` for. +# `exit` requires the watcher to surface and exit; `absorb` requires it to +# survive whole poll cycles - enough to see the new hash, count it stable, and +# reach the stale path. Returns 1 when the watcher does the other thing. +parked_watch_round() { # + local state=$1 fakebin=$2 out=$3 capture=$4 window=$5 mode=$6 pid cycles=0 + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture" \ + FM_FAKE_TMUX_CURRENT_COMMAND=grok \ + FM_FAKE_CREW_STATE='state: paused · source: status-log · parked' \ + FM_WATCH_HANDLING_SUCCESSOR=1 \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" \ + FM_PAUSE_RESURFACE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" >> "$out" & + pid=$! + if [ "$mode" = exit ]; then + wait_for_exit "$pid" 100 || { reap "$pid"; return 1; } + return 0 + fi + while [ "$cycles" -lt 4 ]; do + wait_poll_cycle "$state" "$pid" 300 || { reap "$pid"; return 1; } + cycles=$((cycles + 1)) + done + reap "$pid" + return 0 +} + +# --- a live worker parked on a declared wait: pane churn must not re-alarm ---- +# The 2026-08/09 alarm loop, in both observed forms - a worker parked on the +# CAPTAIN (captain-held, five consecutive alarms) and one parked on the PIPELINE +# (paused:, dozens across one day). pause_state_class deliberately returns `none` +# for either while the agent is still ALIVE, so that a worker genuinely waiting on +# a decision is never silenced; first sight of each distinct stale hash therefore +# reaches surface_nonterminal_stale. An idle parked pane still churns its hash (a +# clock, a token counter), so every tick used to re-enter that first-sight path and +# wake firstmate - the throttle was written by the very wake it should have +# prevented, and the hash-change path cleared it again before it was ever read. +# The contract pinned here: the FIRST sight still surfaces, further sights inside +# PAUSE_RESURFACE_SECS are absorbed, and the window's end still re-surfaces once, +# so a forgotten wait cannot rot invisibly. +test_live_declared_wait_churn_honors_the_resurface_throttle() { + local spec name status_line dir state fakebin out capture_file statusf window key + local sig round wakes bare text throttle replacement + for spec in \ + 'paused-pipeline-churn|paused: waiting on the validation run to finish' \ + 'captain-held-churn|captain-held [key=route]: awaiting the captain on the routing call' + do + name=${spec%%|*}; status_line=${spec#*|} + dir=$(make_case "$name"); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt"; statusf="$state/parked.status" + window="test:fm-parked" + printf 'window=%s\nkind=ship\nharness=grok\nbackend=tmux\n' "$window" > "$state/parked.meta" + printf '%s\n' "$status_line" > "$statusf" + sig=$(seen_sig "$statusf"); printf '%s' "$sig" > "$state/.seen-parked_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + throttle="$state/.paused-resurfaced-$key" + + # First sight of a parked-but-live worker must still surface: the state is + # inconclusive and firstmate has to look at it. + text='parked, elapsed 1s' + printf '%s' "$text" > "$capture_file" + printf '%s' "$(hash_text "$text")" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + parked_watch_round "$state" "$fakebin" "$out" "$capture_file" "$window" exit \ + || fail "[$name] first sight of a parked live worker did not surface" + ack_stopped_cycle "$state" || fail "[$name] could not acknowledge the first surface" + [ -e "$throttle" ] || fail "[$name] the first surface recorded no re-surface throttle" + + # The pane now churns while the SAME declared wait stands, each round fully + # handled as a real supervision turn would. Every one of these used to alarm. + round=2 + while [ "$round" -le 4 ]; do + printf 'parked, elapsed %ss' "$round" > "$capture_file" + parked_watch_round "$state" "$fakebin" "$out" "$capture_file" "$window" absorb \ + || fail "[$name] watcher exited during churn round $round instead of supervising through it" + wakes=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w { n++ } END { print n + 0 }' \ + "$state/.wake-queue" 2>/dev/null || echo 0) + [ "$wakes" -eq 0 ] \ + || fail "[$name] pane churn re-alarmed a parked worker $wakes time(s) inside the re-surface window" + [ -e "$throttle" ] || fail "[$name] pane churn cleared the re-surface throttle" + round=$((round + 1)) + done + + # A direct wait-to-wait transition starts a NEW declaration even though the + # same window remains parked. Its first sight must not inherit the previous + # declaration's throttle, or an unrelated replacement wait can stay silent + # for nearly the whole old cadence window. + case "$name" in + paused-pipeline-churn) replacement='paused: waiting on the replacement validation run' ;; + captain-held-churn) replacement='captain-held [key=release]: awaiting the captain on the release call' ;; + esac + printf '%s\n' "$replacement" >> "$statusf" + sig=$(seen_sig "$statusf"); printf '%s' "$sig" > "$state/.seen-parked_status" + printf 'replacement wait, elapsed 1s' > "$capture_file" + parked_watch_round "$state" "$fakebin" "$out" "$capture_file" "$window" exit \ + || fail "[$name] a replacement declared wait inherited the previous wait's re-surface throttle" + wakes=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w { n++ } END { print n + 0 }' \ + "$state/.wake-queue" 2>/dev/null || echo 0) + bare=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w && $5 == "stale: " w { n++ } END { print n + 0 }' \ + "$state/.wake-queue" 2>/dev/null || echo 0) + [ "$wakes" -eq 1 ] || fail "[$name] replacement declared wait produced $wakes first wakes instead of one" + [ "$bare" -eq 1 ] || fail "[$name] replacement declared wait changed the wake identity: $(cat "$state/.wake-queue")" + ack_stopped_cycle "$state" || fail "[$name] could not acknowledge the replacement wait's first surface" + + printf 'replacement wait, elapsed 2s' > "$capture_file" + parked_watch_round "$state" "$fakebin" "$out" "$capture_file" "$window" absorb \ + || fail "[$name] replacement wait re-alarmed inside its own re-surface window" + wakes=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w { n++ } END { print n + 0 }' \ + "$state/.wake-queue" 2>/dev/null || echo 0) + [ "$wakes" -eq 0 ] || fail "[$name] replacement wait re-alarmed $wakes time(s) inside its own re-surface window" + + # End of the window: the wait must re-surface exactly once, on the same plain + # identity as before, so absorbing churn never becomes silence. + set_mtime "$(( $(date +%s) - 2000 ))" "$throttle" + printf 'parked, elapsed 5s' > "$capture_file" + parked_watch_round "$state" "$fakebin" "$out" "$capture_file" "$window" exit \ + || fail "[$name] a parked worker did not re-surface once its re-surface window elapsed" + wakes=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w { n++ } END { print n + 0 }' \ + "$state/.wake-queue" 2>/dev/null || echo 0) + bare=$(awk -F '\t' -v w="$window" '$3 == "stale" && $4 == w && $5 == "stale: " w { n++ } END { print n + 0 }' \ + "$state/.wake-queue" 2>/dev/null || echo 0) + [ "$wakes" -eq 1 ] || fail "[$name] elapsed re-surface window produced $wakes wakes instead of one" + [ "$bare" -eq 1 ] || fail "[$name] elapsed re-surface changed the wake identity: $(cat "$state/.wake-queue")" + done + pass "a parked live worker surfaces once, absorbs pane churn for the whole re-surface window, then re-surfaces when it elapses" +} + test_secondmate_paused_resurfaces_in_normal_mode() { local dir state fakebin out capture_file statusf window key pane_hash sig pid back home dir=$(make_case secondmate-paused-resurface); state="$dir/state"; fakebin="$dir/fakebin" @@ -3062,6 +3254,8 @@ test_nonterminal_stale_not_working_surfaced test_nonterminal_stale_paused_absorbed_then_resurfaced test_watcher_pause_default_is_2700_and_override_wins test_exited_declared_pause_is_bounded_but_live_gate_surfaces +test_absorbed_replacement_wait_does_not_inherit_the_old_throttle +test_live_declared_wait_churn_honors_the_resurface_throttle test_secondmate_paused_resurfaces_in_normal_mode test_secondmate_nonpaused_stale_remains_suppressed test_remote_secondmate_beacon_is_bounded From 8c4d4cd3c91de63a480fd5393517b0c268f2ea70 Mon Sep 17 00:00:00 2001 From: dnth Date: Sat, 5 Sep 2026 18:22:03 +0800 Subject: [PATCH 2/5] no-mistakes(review): Removed unsupported churn docs and recorded future port --- docs/architecture.md | 10 +--------- 1 file changed, 1 insertion(+), 9 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index a95c74339e8..fc08a39fc4e 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -27,15 +27,6 @@ A concurrent replacement remains armed, every non-merged or invalid observation `bin/fm-pr-lib.sh` owns the notification-marker and retirement-receipt formats plus their strict identity mechanics, [`bin/fm-merge-outcome-lib.sh`](../bin/fm-merge-outcome-lib.sh) owns role-routed publication, the local durable row, and marker ordering, and `bin/fm-watch.sh` owns immediate poll-result delivery and retirement. A merge-outcome row is an ordinary check-kind wake, so the OMP supervision branch's dispatch classifier keeps it main-owned exactly like every other merge-confirmation poll result. No-verb wakes, such as `working:` notes and bare turn-ended signals, are benign only when every referenced task independently has positive evidence that its crew is still working: a currently attributed active no-mistakes step, or an exact busy verdict from the semantic busy-state contract, both read through `bin/fm-crew-state.sh`. -A home that creates `config/turnend-churn-absorb` lets each eligible bare turn-ended task that lacks either authoritative proof use a third form: pane content that changed since the previous poll, compared against the same `state/.hash-*` marker the staleness backbone records, which claims no harness semantics and needs no adapter cooperation. -That form stays opt-in because it infers execution from rendered bytes rather than from a verdict the harness vouches for, so with the flag absent triage behaves exactly as it did before ([`configuration.md`](configuration.md) "Turn-end pane-churn absorb"). -That evidence clears the pane's prior stale classification and wedge-escalation count, then defers such a wake rather than swallowing it, since a crew that has stopped renders nothing further and its now-static pane surfaces through the staleness backbone within a poll or two, even if its final bytes match an earlier stale render. -A wake naming any status file remains governed solely by the strict authoritative proof, and the pane-churn fallback is unavailable to an entire batch that references a secondmate. -An unresolvable endpoint, an ambiguous marker key, a missing or malformed prior hash, a capture that fails or returns empty, an invalid deferral bound or deadline, or an unwritable deferral marker surfaces without clearing prior stale classification. -The deferral is bounded per endpoint by `FM_TURNEND_CHURN_ABSORB_SECS`, tracked in `state/.churn-since-*`, after which the turn-end surfaces and the window restarts. -That bound is load-bearing rather than cosmetic: churn and staleness read the same pane, so a pane that renders continuously - a clock, a spinner, a shell heartbeat, or a harness that leaves a background renderer alive after its agent yields - never reaches the staleness backbone's two-identical-hashes test either, and an unbounded churn absorb would leave a genuinely stopped worker behind such a renderer with no path left to surface it. -If two metadata records derive the same per-window marker key, including two records that name the same endpoint, that marker is not attributable churn evidence for either task, so the bare turn-ended wake surfaces without changing or migrating existing marker state. -A `kind=secondmate` task's status signal is the parent-directed reply stream and is never absorbed as provably working; its bare turn-ended signal is absorbed only by the ordinary authoritative working proof because an active secondmate does not enter the staleness backbone that would resurface deferred pane-churn evidence. A crew that declares `paused:` for a known external wait, or carries a verified `captain-held` transfer, is separately absorbed while idle and re-surfaced only on the longer pause cadence, rather than being treated as a possible wedge. For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint; the pause classification itself is recovered only when the backend confidently reports its agent dead. Live or inconclusive liveness remains fail-open at that initial surface, so a worker genuinely waiting on a decision is never silenced. @@ -379,3 +370,4 @@ Use `/stow` before an intentional reset when the conversation may hold durable k The current watcher reliability work combines always-on bash triage with a durable queue for actionable wakes, generation-bound post-handling acknowledgement, deterministic re-arm recovery after watcher downtime, a race-proof singleton lock, duplicate self-eviction, drain-time liveness assertion, and a self-verifying tracked-child arm wrapper. The presence-gated sub-supervisor (`bin/fm-supervise-daemon.sh`) provides walk-away supervision via the `/afk` skill while reusing the same shared wake classifier as the always-on watcher. +The upstream turn-end pane-churn absorb mode remains a deliberate future port; this fork currently relies on authoritative working proofs and the existing staleness cadence. From 87d17b07ed6ba612536d7618cd3a454fa9bd5478 Mon Sep 17 00:00:00 2001 From: dnth Date: Sat, 5 Sep 2026 18:46:09 +0800 Subject: [PATCH 3/5] no-mistakes(test): Removed duplicate stale escalation configuration documentation --- docs/configuration.md | 1 - 1 file changed, 1 deletion(-) diff --git a/docs/configuration.md b/docs/configuration.md index 2e04f48139a..1febaaa26be 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -704,7 +704,6 @@ FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stal FM_REMOTE_STALE_RECHECK_SECS=60 # seconds between inconclusive remote stale-owner probes during away-mode recovery; invalid values use 60 FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's live-generation state/.turn-ended. marker, or its state/.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait or verified captain-held transfer takes the FM_PAUSE_RESURFACE_SECS recheck below instead FM_PAUSE_RESURFACE_SECS=2700 # seconds between bounded rechecks of a declared external wait or verified captain-held transfer, including a live idle pane after its first inconclusive stale wake and a live busy pane past FM_BUSY_TURN_MAX_SECS; the away-mode daemon uses the same setting, ageing its window against the crew's own latest status line rather than pane busy state -FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless admitted directly to the declared-wait cadence, while a live idle declared wait still surfaces once before that cadence bounds repeats FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added FM_WORKTREE_WRITE_PRUNE='.git node_modules .venv venv __pycache__ .mypy_cache .pytest_cache .ruff_cache .tox target dist build .next .cache vendor' # directory names the wedge detector's task-worktree write probe skips; the default keeps .git out so a supervisor's own read-only git command can never look like crew progress; set it to the empty string to prune nothing, which widens the probe to the whole depth-bounded tree rather than disabling it FM_WORKTREE_WRITE_MAXDEPTH=6 # depth that same probe walks below the recorded worktree; it runs only at the moment a wedge escalation would otherwise fire, never on every poll; no probe knob applies to a secondmate, whose recorded worktree is a provisioned home the probe skips entirely From a2dfdb4b004d692aa5dd84a4757e13004e7468c8 Mon Sep 17 00:00:00 2001 From: dnth Date: Sat, 5 Sep 2026 18:56:01 +0800 Subject: [PATCH 4/5] no-mistakes(document): Document supervision wake-loop behavior and regression entry points --- docs/verification/supervision.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index fa64afae168..2909a50ced0 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -394,9 +394,11 @@ tests/fm-pi-primary-types.test.sh tests/fm-watcher-lock.test.sh tests/fm-watch-arm.test.sh tests/fm-watch-recovery-loop.test.sh +tests/fm-watch-triage.test.sh tests/fm-wake-drain-unread-status.test.sh tests/fm-wake-queue.test.sh tests/fm-wake-daemon-lifecycle-e2e.test.sh +tests/fm-omp-branch-supervision.test.sh tests/fm-subagent-pretool-check.test.sh tests/fm-claude-stop-autoarm.test.sh tests/fm-turnend-guard.test.sh From 4bd7c324ec61f3b75b83127d00cde87bd749b95a Mon Sep 17 00:00:00 2001 From: dnth Date: Sat, 5 Sep 2026 19:53:44 +0800 Subject: [PATCH 5/5] no-mistakes(document): Remove unsupported turn-end churn documentation --- docs/architecture.md | 1 - 1 file changed, 1 deletion(-) diff --git a/docs/architecture.md b/docs/architecture.md index fc08a39fc4e..9c02a9efaf1 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -370,4 +370,3 @@ Use `/stow` before an intentional reset when the conversation may hold durable k The current watcher reliability work combines always-on bash triage with a durable queue for actionable wakes, generation-bound post-handling acknowledgement, deterministic re-arm recovery after watcher downtime, a race-proof singleton lock, duplicate self-eviction, drain-time liveness assertion, and a self-verifying tracked-child arm wrapper. The presence-gated sub-supervisor (`bin/fm-supervise-daemon.sh`) provides walk-away supervision via the `/afk` skill while reusing the same shared wake classifier as the always-on watcher. -The upstream turn-end pane-churn absorb mode remains a deliberate future port; this fork currently relies on authoritative working proofs and the existing staleness cadence.