diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f3e5adcf030..ad7bdc7d6c9 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,7 +31,7 @@ jobs: name: Lint ${{ matrix.partition }} runs-on: ubuntu-latest # Normal tier (see the timeout policy above). - timeout-minutes: 30 + timeout-minutes: 60 strategy: fail-fast: false matrix: @@ -92,7 +92,7 @@ jobs: name: Behavior portable parallel 1 runs-on: ubuntu-latest # Normal tier (see the timeout policy above). - timeout-minutes: 30 + timeout-minutes: 60 steps: - uses: actions/checkout@v6 with: @@ -112,10 +112,13 @@ jobs: set -eu npm install -g tasks-axi tasks-axi --version + # Pinned for the same reason as the serial-shard install below: Pi 1.0.0 + # moved the interactive TUI onto the alternate screen and breaks the live + # extension E2Es. Unpin once those E2Es are ported to the 1.0.0 renderer. - name: Install the Pi package for the Pi extension tests run: | set -eu - npm install -g @earendil-works/pi-coding-agent + npm install -g @earendil-works/pi-coding-agent@0.99.2 npm ls -g --depth 0 @earendil-works/pi-coding-agent - name: Run portable parallel shard 1 run: | @@ -136,7 +139,7 @@ jobs: name: Behavior portable parallel 2 runs-on: ubuntu-latest # Normal tier (see the timeout policy above). - timeout-minutes: 30 + timeout-minutes: 60 steps: - uses: actions/checkout@v6 with: @@ -180,7 +183,7 @@ jobs: name: Behavior portable serial ${{ matrix.shard }} runs-on: ubuntu-latest # Normal tier (see the timeout policy above). - timeout-minutes: 30 + timeout-minutes: 60 strategy: # Every shard reports so one failure never hides another shard's result. fail-fast: false @@ -216,10 +219,15 @@ jobs: # The Pi extension tests read the installed Pi package's own types and # runtime, so without it they gate-skip and pass silently. It is a public # npm package and needs no credential, so CI can hold the real thing. + # The version is pinned like the other tool installs above: Pi 1.0.0 + # moved the interactive TUI onto the alternate screen, which drops the + # earliest restored-transcript rows out of every pane capture and fails + # the live calm/extension E2Es (serial 6, runs 36929988626, 36938007991, + # 36945567914). Unpin once those E2Es are ported to the 1.0.0 renderer. - name: Install the Pi package for the Pi extension tests run: | set -eu - npm install -g @earendil-works/pi-coding-agent + npm install -g @earendil-works/pi-coding-agent@0.99.2 npm ls -g --depth 0 @earendil-works/pi-coding-agent - name: Run portable serial shard ${{ matrix.shard }} env: @@ -413,7 +421,7 @@ jobs: name: Stock macOS Bash snapshot compatibility runs-on: macos-latest # Normal tier (see the timeout policy above). - timeout-minutes: 30 + timeout-minutes: 60 steps: - uses: actions/checkout@v6 - name: Run snapshot consumers with stock Bash diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index e3fe51fd2ae..864b2c857df 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -348,9 +348,10 @@ INBOX_DIR=$(shell_quote "$STATE/$ID.inbox") # The receive-and-ack half of the steering-inbox contract, included in every # scaffold kind. The record format, doorbell line, and re-ring ladder are -# owned by bin/fm-task-inbox-lib.sh; the doorbell itself is self-describing, -# so this section is reinforcement for the natural-checkpoint habit, not the -# only carrier of the instruction. +# owned by bin/fm-task-inbox-lib.sh. The doorbell names the inbox as +# "$FM_TASK_INBOX", which bin/fm-spawn.sh exports into every launch; the full +# path here remains the fallback for a worker launched without that export, +# plus the natural-checkpoint habit. IFS= read -r -d '' INBOX_SECTION < (or a child process that # inherits it) carries the override there, and a lookup that honored it # would find this directory again and never run the repository's own -# hook - a skipped pre-push guard. A lookup that fails exits nonzero -# rather than skipping the repository's hook. Does not touch the +# hook - a skipped pre-push guard. An empty core.hooksPath means no +# repository hook, as in plain git; any other failed lookup exits +# nonzero rather than skipping the repository's hook. Does not touch the # project's git config; the caller prefixes the pane with # GIT_CONFIG_COUNT / GIT_CONFIG_KEY_0 / GIT_CONFIG_VALUE_0. # @@ -153,14 +154,21 @@ write_executable() { # is the other environment channel that can carry this directory as # core.hooksPath; only the repository's config files name its own hooks. Skip # when the lookup still names this launch's own hooks dir, meaning those files -# point here, so the wrapper cannot recurse into itself. +# point here, so the wrapper cannot recurse into itself. An empty +# core.hooksPath makes that lookup fail, but plain git reads it as "no hooks", +# so the wrapper runs none; any other failure reruns the lookup to show git's +# error and refuses. runtime_chain_body() { local ours=$1 cat </dev/null) || { + if hooks_path=\$(unset GIT_CONFIG_PARAMETERS; git config --get --type=path core.hooksPath 2>/dev/null) && [ -z "\$hooks_path" ]; then + exit 0 + fi + (unset GIT_CONFIG_PARAMETERS; git rev-parse --path-format=absolute --git-path hooks >/dev/null) echo "fm-git-strip-ai-trailers: cannot resolve this repository's hooks directory; refusing to skip its \$name hook" >&2 exit 1 } diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 2fafee24428..f98564a280a 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -324,7 +324,9 @@ # pins to 1 with a literal assignment so it survives the cleared environment # even on a host that never had it set. # An enabled task trace also retains TRACEPARENT. Explicit Firstmate launch -# assignments still apply inside the filtered environment. Raw commands must +# assignments still apply inside the filtered environment, including the +# FM_TASK_INBOX export every launch carries (the absolute state/.inbox +# path the steering doorbell names). Raw commands must # be POSIX sh compatible under this opt-in; the absent-file path is unchanged. # This is an exec environment boundary, not a sandbox for the pane's startup # shell, credential files, same-user processes, or later shell initialization. @@ -5773,6 +5775,12 @@ fi if [ "$LAVISH_AXI_HOST_CONFIG_PRESENT" = 1 ]; then LAUNCH="export LAVISH_AXI_HOST=$(shell_quote "$LAVISH_AXI_HOST"); $LAUNCH" fi +# Every launch also exports the absolute path of this task's steering inbox, so +# the constant doorbell line (bin/fm-task-inbox-lib.sh) can name +# "$FM_TASK_INBOX" instead of a path that grows with the home's depth. Like the +# kill switch below it is an export statement, so it survives a compound raw +# launch and the launch-env-allowlist `env -i` wrapper. +LAUNCH="export FM_TASK_INBOX=$(shell_quote "$STATE_REAL/$ID.inbox"); $LAUNCH" LAUNCH="export COMPACT_ADVISER_DISABLE=1; $LAUNCH" # When the live-harness gate has exported DISABLE_AUTOUPDATER into this spawn's # own environment, carry it into the launch command text so Claude Code's diff --git a/bin/fm-task-inbox-lib.sh b/bin/fm-task-inbox-lib.sh index 95f661062d8..aa95cdcb7ed 100644 --- a/bin/fm-task-inbox-lib.sh +++ b/bin/fm-task-inbox-lib.sh @@ -59,7 +59,7 @@ # crash or marker failure may produce a rare duplicate rather than silently lose # a wake. # -# Inbox paths containing bytes outside printable ASCII are unsupported. The +# Inbox names containing bytes outside printable ASCII are unsupported. The # doorbell refuses them rather than sending terminal control bytes to a pane. # # fm_task_inbox_ring requires bin/fm-backend.sh's dispatch (sourced below); the @@ -252,22 +252,30 @@ fm_task_inbox_body() { # } # The constant self-describing doorbell line for the inbox containing a record. -# Self-describing on purpose: a worker whose brief predates the inbox contract -# still receives the complete instruction in the line itself. The leading `: ` -# is the POSIX shell no-op, so the same line typed into a pane whose agent has -# exited (a bare shell) runs nothing; see the dead-pane note in the header. -# A non-printable path fails without output so terminal controls never reach -# the pane's line discipline. +# It names the inbox by the literal "$FM_TASK_INBOX", which bin/fm-spawn.sh +# exports into every launch as the inbox's absolute path, so the worker can +# resolve it from its own environment even after losing its brief context. +# The short `.inbox` name follows as the fallback for a worker launched +# before that export, whose brief carries the full path (bin/fm-dod-lib.sh +# role contract, bin/fm-brief.sh inbox section). No absolute path is printed, +# so the line's length never grows with the home's depth: a long line wraps +# past what a harness composer read can prove, and a Herdr submit then reports +# it did not reach the pane on every re-ring. The leading `: ` is the POSIX +# shell no-op, so the same line typed into a pane whose agent has exited (a +# bare shell) runs nothing; see the dead-pane note in the header. A +# non-printable inbox name fails without output so terminal controls never +# reach the pane's line discipline. fm_task_inbox_doorbell_line() { # - local dir=${1%/*} abs quoted LC_ALL=C + local dir=${1%/*} abs name quoted LC_ALL=C abs=$(cd "$dir" 2>/dev/null && pwd) || abs=$dir abs=${abs%/handled} - case "$abs" in - *[![:print:]]*) return 1 ;; + name=${abs##*/} + case "$name" in + ''|*[![:print:]]*) return 1 ;; esac - quoted=$(printf '%s' "$abs" | sed "s/'/'\\\\''/g") - printf ": Firstmate instruction waiting: list '%s'/*.msg and, in numeric order, read and act on each, then mv each handled file to '%s'/handled/." \ - "$quoted" "$quoted" + quoted=$(printf '%s' "$name" | sed "s/'/'\\\\''/g") + printf ": Firstmate instruction waiting: list \"\$FM_TASK_INBOX\"/*.msg in your '%s' steering inbox, read and act on each in numeric order, then mv each into its handled/." \ + "$quoted" } # Ring the doorbell, best-effort: one endpoint-liveness pre-check, one advisory diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index 7f0ef959576..328f112e689 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -2383,6 +2383,12 @@ require_exclusive_worktree_slot_record() { local record_meta=$1 record_id=$2 record_state=$3 worktree=$4 local slot state_dir other other_id field other_path other_slot slot=$(canonical_existing_dir "$worktree") || return 0 + # A slot whose owner claim names another task was reassigned, so this record's + # teardown is records-only and touches nothing under it; another record naming + # the slot is then no hazard, and refusing would strand this stale record and + # block the claimant's own teardown behind it. + fm_treehouse_slot_owner_state "$slot" "$record_id" + [ "$FM_TREEHOUSE_SLOT_OWNER" != other ] || return 0 collect_local_firstmate_states "$record_state" || return 1 for state_dir in "${TREEHOUSE_OWNER_STATES[@]}"; do for other in "$state_dir"/*.meta; do diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 991301de1ac..5acc4e39fab 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -71,9 +71,9 @@ # --per-script-timeout-secs N # terminate a script that runs longer than N seconds and # record it as exit 124 (0 disables, the default). The -# --changed applies 1500s automatically: no measured script -# approaches it, so it only converts a HUNG -# script into a bounded failure. --max-wall-ms is checked +# --changed applies 1500s automatically, above the current +# slowest CI hint with margin; exceeding it becomes a bounded +# failure, not proof of a hang. --max-wall-ms is checked # after the run and so cannot catch a hang on its own. # External interruption cleanup is outside this runner's # guarantee; configured per-script bounds remain authoritative. @@ -135,6 +135,10 @@ # split it across separate runners, so two of its stateful scripts still never # share a machine. This script owns : a lane whose disagrees with the # configured shard count is refused, so a CI matrix cannot silently drop a shard. +# --check-coverage also reports serial_max_ms (largest packed hint sum, including +# default weights) and serial_budget_ms (the 20-minute packing target), refusing +# a split above that target. Neither figure is an execution timeout or proof of +# observed headroom: refresh growing files from CI measurements. # --changed is conservative: it over-selects related families rather than # under-selecting, and never expands to the complete suite unless --all. The one # place it is deliberately narrow is a bin/ path with no curated family: a test @@ -183,13 +187,11 @@ MAX_WALL_MS= PER_SCRIPT_TIMEOUT_SECS=0 # Bound applied automatically on the automatic --changed path, derived from # measured healthy runtimes with margin rather than picked: the slowest measured -# script is tests/fm-watch-triage.test.sh in the watcher-wake-lock family, at -# about 434s alone and about 698s under CI load (the hint table below records -# that loaded figure), and the slowest script in a runner-file changed selection -# is tests/fm-calm-pi-extension.test.sh at 77s once its Chrome reap terminates. -# 1500s keeps every measured script under the bound with roughly 2.1x headroom -# over the slowest loaded measurement, and it stays under the 30-minute normal -# CI tier so a wedged script fails here, with its output, before the job cap +# script is tests/fm-watch-triage.test.sh at about 1075s under CI load (the hint +# table below records that loaded figure). 1500s keeps every measured script +# under the bound with roughly 1.4x headroom over the slowest loaded measurement, +# and it stays under the 60-minute normal CI tier so a wedged script fails here, +# with its output, before the job cap # cancels the lane. It is a guard, not a speed control: a HUNG script becomes a # bounded failure instead of an unbounded suite, which is the shape that # silently outruns a caller's invocation budget. @@ -199,10 +201,14 @@ CHANGED_DEFAULT_TIMEOUT_SECS=1500 # One owner: CI lane names carry this count and are refused when they disagree. PORTABLE_SERIAL_SHARDS=9 -# Balance hint for a portable-serial script with no measured duration, close to -# the measured per-script mean so a newly added test neither starves nor -# overloads the shard it lands in. -PORTABLE_SERIAL_DEFAULT_WEIGHT_MS=27000 +# Conservative balance hint for a portable-serial script with no measurement. +# Rounded above the current CI mean, including the capability-skipped scripts. +PORTABLE_SERIAL_DEFAULT_WEIGHT_MS=45000 + +# Packing target, not an execution timeout: leave at least forty minutes of the +# normal CI tier for setup and runtime variance. --check-coverage refuses a +# modeled serial shard above this target; refresh hints or rebalance instead. +PORTABLE_SERIAL_MAX_WEIGHT_MS=1200000 # Largest share of the serial lane allowed to run on the default weight above. # Hints are what keep the shards balanced, so once too much of the lane is @@ -516,30 +522,30 @@ EOF # refresh procedure are owned by docs/fm-test-portable-shards.md. portable_parallel_weight_hints() { cat <<'EOF' -tests/fm-arm-pretool-check.test.sh 30898 -tests/fm-backend-herdr.test.sh 22144 -tests/fm-brief.test.sh 1625 -tests/fm-captain-hold-lifecycle.test.sh 296481 -tests/fm-cd-pretool-check.test.sh 16964 -tests/fm-composer-ghost.test.sh 2120 -tests/fm-composer-lib.test.sh 4798 -tests/fm-crew-state.test.sh 11557 -tests/fm-ensure-agents-md.test.sh 901 -tests/fm-grok-harness.test.sh 6563 -tests/fm-herdr-lab.test.sh 9800 -tests/fm-lint.test.sh 164262 -tests/fm-pi-primary-types.test.sh 8624 -tests/fm-pr-merge.test.sh 111145 -tests/fm-review-diff.test.sh 2747 -tests/fm-send-popup-settle.test.sh 4939 -tests/fm-send-settle.test.sh 2051 -tests/fm-send-strict.test.sh 3861 -tests/fm-spawn-batch.test.sh 2265 -tests/fm-supervision-instructions.test.sh 297 -tests/fm-test-run.test.sh 92944 -tests/fm-tmux-submit-busy.test.sh 2477 -tests/fm-transition-lib.test.sh 99 -tests/fm-x-mode.test.sh 31870 +tests/fm-arm-pretool-check.test.sh 33778 +tests/fm-backend-herdr.test.sh 36331 +tests/fm-brief.test.sh 10594 +tests/fm-captain-hold-lifecycle.test.sh 343658 +tests/fm-cd-pretool-check.test.sh 16801 +tests/fm-composer-ghost.test.sh 2292 +tests/fm-composer-lib.test.sh 9521 +tests/fm-crew-state.test.sh 82058 +tests/fm-ensure-agents-md.test.sh 895 +tests/fm-grok-harness.test.sh 7666 +tests/fm-herdr-lab.test.sh 18325 +tests/fm-lint.test.sh 252498 +tests/fm-pi-primary-types.test.sh 5426 +tests/fm-pr-merge.test.sh 300199 +tests/fm-review-diff.test.sh 4134 +tests/fm-send-popup-settle.test.sh 6624 +tests/fm-send-settle.test.sh 2310 +tests/fm-send-strict.test.sh 4804 +tests/fm-spawn-batch.test.sh 2987 +tests/fm-supervision-instructions.test.sh 809 +tests/fm-test-run.test.sh 156781 +tests/fm-tmux-submit-busy.test.sh 2600 +tests/fm-transition-lib.test.sh 101 +tests/fm-x-mode.test.sh 29896 EOF } @@ -562,17 +568,19 @@ portable_parallel_lane_weight() { # workflow step moved with it. list_portable_parallel_1() { cat <<'EOF' -tests/fm-lint.test.sh tests/fm-pr-merge.test.sh -tests/fm-test-run.test.sh +tests/fm-lint.test.sh +tests/fm-backend-herdr.test.sh +tests/fm-x-mode.test.sh tests/fm-cd-pretool-check.test.sh -tests/fm-pi-primary-types.test.sh -tests/fm-grok-harness.test.sh tests/fm-composer-lib.test.sh +tests/fm-send-popup-settle.test.sh +tests/fm-pi-primary-types.test.sh tests/fm-review-diff.test.sh -tests/fm-tmux-submit-busy.test.sh -tests/fm-composer-ghost.test.sh -tests/fm-brief.test.sh +tests/fm-send-settle.test.sh +tests/fm-ensure-agents-md.test.sh +tests/fm-supervision-instructions.test.sh +tests/fm-transition-lib.test.sh EOF } @@ -580,18 +588,16 @@ EOF list_portable_parallel_2() { cat <<'EOF' tests/fm-captain-hold-lifecycle.test.sh -tests/fm-x-mode.test.sh -tests/fm-arm-pretool-check.test.sh -tests/fm-backend-herdr.test.sh +tests/fm-test-run.test.sh tests/fm-crew-state.test.sh +tests/fm-arm-pretool-check.test.sh tests/fm-herdr-lab.test.sh -tests/fm-send-popup-settle.test.sh +tests/fm-brief.test.sh +tests/fm-grok-harness.test.sh tests/fm-send-strict.test.sh tests/fm-spawn-batch.test.sh -tests/fm-send-settle.test.sh -tests/fm-ensure-agents-md.test.sh -tests/fm-supervision-instructions.test.sh -tests/fm-transition-lib.test.sh +tests/fm-tmux-submit-busy.test.sh +tests/fm-composer-ghost.test.sh EOF } @@ -675,193 +681,214 @@ list_portable_serial() { # Measured portable-serial script durations in milliseconds, from the CI timing # artifacts recorded in docs/fm-test-portable-shards.md. Each value is the -# slowest successful sample in the referenced complete/partial CI runs, rather -# than only on the fastest one measured. These are balance hints only: the shard +# slowest successful sample in the referenced complete/partial CI runs, with +# the version-specific host and native-Windows exceptions documented there. +# These are balance hints only: the shard # partition stays complete and disjoint whatever they say, so a stale hint costs # balance rather than coverage. That doc owns the refresh procedure. portable_serial_weight_hints() { cat <<'EOF' -tests/fm-afk-contract.test.sh 15645 -tests/fm-afk-inject-e2e.test.sh 35889 -tests/fm-afk-pi-herdr-return-e2e.test.sh 45 -tests/fm-afk-return.test.sh 20385 -tests/fm-agy-harness.test.sh 47933 -tests/fm-agy-signals-live-e2e.test.sh 49 -tests/fm-ask-user-authority.test.sh 131 -tests/fm-backend-cmux-smoke.test.sh 33 -tests/fm-backend-cmux.test.sh 3498 -tests/fm-backend-orca.test.sh 23381 -tests/fm-backend-tmux-smoke.test.sh 363 -tests/fm-backend-zellij-smoke.test.sh 21 -tests/fm-backend-zellij.test.sh 9064 -tests/fm-backend.test.sh 21658 -tests/fm-backlog-atomicity.test.sh 196948 -tests/fm-backlog-handoff.test.sh 51990 -tests/fm-backlog-read-bound.test.sh 24288 -tests/fm-bearings-board-lavish-live-e2e.test.sh 48 -tests/fm-bearings-board-render.test.sh 12591 -tests/fm-bearings-board.test.sh 36490 -tests/fm-bearings-snapshot.test.sh 171176 -tests/fm-bootstrap-network-parallel.test.sh 9539 -tests/fm-bootstrap.test.sh 46634 -tests/fm-branch-supervision.test.sh 8915 -tests/fm-busy-adapter-wiring.test.sh 27817 -tests/fm-busy-state.test.sh 2990 -tests/fm-calm-claude-mod-live-e2e.test.sh 46 -tests/fm-calm-claude-mod-plugin.test.sh 172 -tests/fm-calm-claude-mod.test.sh 1252 -tests/fm-calm-pi-extension.test.sh 45128 -tests/fm-check-unregister.test.sh 464 -tests/fm-ci-workflow.test.sh 2073 -tests/fm-classify-corr-token.test.sh 49294 -tests/fm-classify-decision-key.test.sh 3336 -tests/fm-claude-stop-autoarm-live-e2e.test.sh 45 -tests/fm-claude-stop-autoarm.test.sh 60797 -tests/fm-claude-trust.test.sh 10410 -tests/fm-cmux-claude-composer-live-e2e.test.sh 47 -tests/fm-codex-continuity-live-e2e.test.sh 71 -tests/fm-codex-hook-layer-live-e2e.test.sh 47 -tests/fm-composer-codex-idle-live-e2e.test.sh 229 -tests/fm-composer-matrix-live-e2e.test.sh 47 -tests/fm-contributions.test.sh 35676 -tests/fm-control-relaunch.test.sh 137013 -tests/fm-control.test.sh 39524 -tests/fm-cursor-harness.test.sh 30212 -tests/fm-cursor-primary-live-e2e.test.sh 72 -tests/fm-cursor-primary.test.sh 52269 -tests/fm-daemon.test.sh 27262 -tests/fm-dispatch-resolve.test.sh 4397 -tests/fm-documentation-audiences.test.sh 847 -tests/fm-dod-lib.test.sh 4000 -tests/fm-extension-binding.test.sh 9053 -tests/fm-fleet-snapshot-view.test.sh 17465 -tests/fm-fleet-sync.test.sh 35983 -tests/fm-forge-detect.test.sh 160 -tests/fm-gate-refuse.test.sh 5328 -tests/fm-gemini-harness.test.sh 938 -tests/fm-gitignore-config.test.sh 58 -tests/fm-gotmp.test.sh 1320 -tests/fm-grok-continuity-live-e2e.test.sh 45 -tests/fm-grok-stop-live-e2e.test.sh 46 -tests/fm-guard-stale-banner.test.sh 14968 -tests/fm-harness-adapter-instructions-live-e2e.test.sh 48 -tests/fm-harness-adapter-references.test.sh 83 -tests/fm-harness-liveness-drift-live-e2e.test.sh 881 -tests/fm-harness-precedence.test.sh 3661 -tests/fm-herdr-pi-stale-registration-live-e2e.test.sh 47 -tests/fm-herdr-session-cleanup.test.sh 6828 -tests/fm-herdr-submit-confirm-live-e2e.test.sh 46 -tests/fm-herdr-version-floor-live-e2e.test.sh 72 -tests/fm-home-summary-refresh.test.sh 37264 -tests/fm-inactive-reconcile.test.sh 53178 -tests/fm-kimi-harness.test.sh 19151 -tests/fm-lint-workflows.test.sh 785 -tests/fm-live-gate.test.sh 1755 -tests/fm-mail-check.test.sh 9162 -tests/fm-mail.test.sh 9703 -tests/fm-muse-harness.test.sh 40970 -tests/fm-muse-signals-live-e2e.test.sh 77 -tests/fm-nm-test-contract.test.sh 128 -tests/fm-no-mistakes-required.test.sh 247 -tests/fm-omp-harness.test.sh 47734 -tests/fm-omp-primary-live-e2e.test.sh 46 -tests/fm-on.test.sh 11001 -tests/fm-opencode-primary-live-e2e.test.sh 48 -tests/fm-operational-input.test.sh 221 -tests/fm-peek-remote.test.sh 964 -tests/fm-pending-reply.test.sh 28255 -tests/fm-pi-branch-extension.test.sh 60394 -tests/fm-pi-branch-live-e2e.test.sh 72 -tests/fm-pi-branch-responsiveness-live-e2e.test.sh 13121 -tests/fm-pi-codex-native.test.sh 46 -tests/fm-pi-primary-live-e2e.test.sh 47 -tests/fm-pi-watch-extension.test.sh 50637 +tests/fm-afk-contract.test.sh 11101 +tests/fm-afk-inject-e2e.test.sh 41958 +tests/fm-afk-pi-herdr-return-e2e.test.sh 52 +tests/fm-afk-return.test.sh 47380 +tests/fm-agy-harness.test.sh 50959 +tests/fm-agy-signals-live-e2e.test.sh 53 +tests/fm-ask-user-authority.test.sh 171 +tests/fm-backend-cmux-smoke.test.sh 34 +tests/fm-backend-cmux.test.sh 3754 +tests/fm-backend-orca.test.sh 27102 +tests/fm-backend-tmux-smoke.test.sh 291 +tests/fm-backend-zellij-smoke.test.sh 23 +tests/fm-backend-zellij.test.sh 10453 +tests/fm-backend.test.sh 23932 +tests/fm-backlog-atomicity.test.sh 219379 +tests/fm-backlog-handoff.test.sh 57458 +tests/fm-backlog-read-bound.test.sh 24743 +tests/fm-bearings-board-lavish-live-e2e.test.sh 51 +tests/fm-bearings-board-render.test.sh 15612 +tests/fm-bearings-board.test.sh 40817 +tests/fm-bearings-snapshot.test.sh 186219 +tests/fm-bootstrap-network-parallel.test.sh 30424 +tests/fm-bootstrap.test.sh 50965 +tests/fm-branch-supervision.test.sh 22979 +tests/fm-busy-adapter-wiring.test.sh 31642 +tests/fm-busy-state.test.sh 3185 +tests/fm-calm-claude-mod-live-e2e.test.sh 47 +tests/fm-calm-claude-mod-plugin.test.sh 77 +tests/fm-calm-claude-mod.test.sh 2527 +tests/fm-calm-pi-extension.test.sh 56463 +tests/fm-calm-pi-queue-retention-live-e2e.test.sh 1345 +tests/fm-check-unregister.test.sh 469 +tests/fm-ci-workflow.test.sh 5833 +tests/fm-classify-corr-token.test.sh 23085 +tests/fm-classify-decision-key.test.sh 4362 +tests/fm-claude-stop-autoarm-live-e2e.test.sh 73 +tests/fm-claude-stop-autoarm.test.sh 61189 +tests/fm-claude-trust.test.sh 12010 +tests/fm-cmux-claude-composer-live-e2e.test.sh 77 +tests/fm-codex-continuity-live-e2e.test.sh 108 +tests/fm-codex-hook-layer-live-e2e.test.sh 108 +tests/fm-composer-codex-idle-live-e2e.test.sh 77 +tests/fm-composer-matrix-live-e2e.test.sh 51 +tests/fm-contributions.test.sh 140911 +tests/fm-control-relaunch.test.sh 114115 +tests/fm-control.test.sh 72794 +tests/fm-cursor-harness.test.sh 30088 +tests/fm-cursor-primary-live-e2e.test.sh 75 +tests/fm-cursor-primary.test.sh 69845 +tests/fm-daemon.test.sh 33606 +tests/fm-devin-harness.test.sh 3725 +tests/fm-devin-signals-live-e2e.test.sh 49 +tests/fm-dispatch-resolve.test.sh 10051 +tests/fm-documentation-audiences.test.sh 1301 +tests/fm-dod-lib.test.sh 2035 +tests/fm-extension-binding.test.sh 11105 +tests/fm-fleet-ledger.test.sh 19980 +tests/fm-fleet-snapshot-view.test.sh 23334 +tests/fm-fleet-sync.test.sh 40541 +tests/fm-forge-detect.test.sh 193 +tests/fm-fork-free-helpers.test.sh 746 +tests/fm-gate-refuse.test.sh 9953 +tests/fm-gemini-harness.test.sh 947 +tests/fm-git-strip-ai-trailers.test.sh 2067 +tests/fm-gitignore-config.test.sh 59 +tests/fm-gotmp.test.sh 1509 +tests/fm-grok-continuity-live-e2e.test.sh 46 +tests/fm-grok-stop-live-e2e.test.sh 48 +tests/fm-guard-stale-banner.test.sh 17234 +tests/fm-harness-adapter-instructions-live-e2e.test.sh 72 +tests/fm-harness-adapter-references.test.sh 64 +tests/fm-harness-liveness-drift-live-e2e.test.sh 1309 +tests/fm-harness-precedence.test.sh 4083 +tests/fm-herdr-pi-stale-registration-live-e2e.test.sh 55 +tests/fm-herdr-session-cleanup.test.sh 7425 +tests/fm-herdr-submit-confirm-live-e2e.test.sh 51 +tests/fm-herdr-version-floor-live-e2e.test.sh 50 +tests/fm-home-summary-refresh.test.sh 37057 +tests/fm-host-mirror-live-e2e.test.sh 79 +tests/fm-host-mirror.test.sh 11587 +tests/fm-inactive-reconcile.test.sh 60823 +tests/fm-inbox.test.sh 6062 +tests/fm-jev-mem-guard.test.sh 336 +tests/fm-kimi-harness.test.sh 58917 +tests/fm-launch-prompt-signals-live-e2e.test.sh 50 +tests/fm-lint-workflows.test.sh 872 +tests/fm-live-gate.test.sh 7452 +tests/fm-live-lab-up-mate.test.sh 17363 +tests/fm-live-lab.test.sh 79639 +tests/fm-mail-check.test.sh 7524 +tests/fm-mail.test.sh 9684 +tests/fm-muse-harness.test.sh 46548 +tests/fm-muse-signals-live-e2e.test.sh 52 +tests/fm-nm-test-contract.test.sh 853 +tests/fm-no-mistakes-required.test.sh 270 +tests/fm-omp-harness.test.sh 63796 +tests/fm-omp-primary-live-e2e.test.sh 74 +tests/fm-on.test.sh 11473 +tests/fm-opencode-primary-live-e2e.test.sh 47 +tests/fm-operational-input.test.sh 2404 +tests/fm-peek-remote.test.sh 1082 +tests/fm-pending-reply.test.sh 41090 +tests/fm-pi-branch-extension.test.sh 77218 +tests/fm-pi-branch-live-e2e.test.sh 48 +tests/fm-pi-branch-responsiveness-live-e2e.test.sh 12834 +tests/fm-pi-codex-native.test.sh 75 +tests/fm-pi-primary-live-e2e.test.sh 72 +tests/fm-pi-watch-extension.test.sh 56515 tests/fm-pi-windows-shell-invocation.test.sh 5121 -tests/fm-pr-check-security.test.sh 226546 -tests/fm-pr-reviewers.test.sh 273 -tests/fm-pr-state-live-e2e.test.sh 45 -tests/fm-pr-state.test.sh 531 -tests/fm-procevent-quota.test.sh 1900 -tests/fm-procevent-when.test.sh 23805 -tests/fm-procevent.test.sh 221745 -tests/fm-project-origin.test.sh 136 -tests/fm-public-followup.test.sh 153508 -tests/fm-quota-array-dispatch-live-e2e.test.sh 71 -tests/fm-quota-choose.test.sh 1484 -tests/fm-remote-backlog-handoff.test.sh 73123 -tests/fm-remote-doctor.test.sh 13889 -tests/fm-remote-entrypoint.test.sh 108 -tests/fm-remote-herdr-guard.test.sh 3044 -tests/fm-remote-job-orphan-reap.test.sh 2905 -tests/fm-remote-job.test.sh 59354 -tests/fm-remote-reply.test.sh 118669 -tests/fm-remote-secondmate-lifecycle-e2e.test.sh 241208 -tests/fm-remote-secondmate-parent-binding.test.sh 32176 -tests/fm-remote-secondmate-trace-context.test.sh 59689 -tests/fm-remote-transport-lanes.test.sh 62635 -tests/fm-rovo-harness.test.sh 14322 -tests/fm-rovo-signals-live-e2e.test.sh 48 -tests/fm-secondmate-harness.test.sh 163801 -tests/fm-secondmate-lifecycle-e2e.test.sh 9633 -tests/fm-secondmate-liveness.test.sh 10402 -tests/fm-secondmate-reconcile.test.sh 97544 -tests/fm-secondmate-restart.test.sh 44488 -tests/fm-secondmate-safety.test.sh 127260 -tests/fm-secondmate-sync.test.sh 54502 -tests/fm-send-agy-confirm.test.sh 3983 -tests/fm-send-inbox-doorbell-live-e2e.test.sh 46 -tests/fm-send-inbox.test.sh 38632 -tests/fm-send-remote-delivery.test.sh 27717 -tests/fm-send-resolve-key.test.sh 28685 -tests/fm-send-secondmate-marker-herdr-e2e.test.sh 52 -tests/fm-send-secondmate-marker.test.sh 5309 -tests/fm-session-lock-ancestry.test.sh 2857 -tests/fm-session-start.test.sh 179350 -tests/fm-sessionstart-hook-live-e2e.test.sh 97 -tests/fm-sessionstart-instruction-refresh-live-e2e.test.sh 46 -tests/fm-sessionstart-nudge.test.sh 66247 -tests/fm-shared-captain-inheritance.test.sh 5687 -tests/fm-spawn-dispatch-profile.test.sh 138433 -tests/fm-spawn-pool-base-freshen.test.sh 62249 -tests/fm-spawn-worktree-settle.test.sh 8482 -tests/fm-startup-memory-budget.test.sh 7392 -tests/fm-startup-network.test.sh 61336 -tests/fm-stat-shadowing.test.sh 48 -tests/fm-stow-cascade.test.sh 3022 -tests/fm-subagent-pretool-check.test.sh 949 -tests/fm-supervision-events.test.sh 659 -tests/fm-supervision-host-live-e2e.test.sh 50 -tests/fm-supervision-host.test.sh 41512 -tests/fm-tangle-guard.test.sh 7470 -tests/fm-task-delivery.test.sh 19784 -tests/fm-task-inbox.test.sh 30004 -tests/fm-tasks-axi.test.sh 1953 -tests/fm-teardown-endpoint-safety.test.sh 33210 -tests/fm-teardown.test.sh 145174 -tests/fm-test-fixture-cleanup.test.sh 937 -tests/fm-test-fixtures.test.sh 1562 -tests/fm-test-isolation-proof.test.sh 2692 -tests/fm-timeout-lib.test.sh 8541 -tests/fm-tmux-agent-liveness.test.sh 1953 -tests/fm-tool-update-check.test.sh 13832 -tests/fm-trace-context-lib.test.sh 227 -tests/fm-trace-context-spawn.test.sh 49071 -tests/fm-turnend-foreign-owner-arm-fix.test.sh 2397 -tests/fm-turnend-guard.test.sh 33450 -tests/fm-update.test.sh 11572 -tests/fm-vendor-auth-probe.test.sh 45255 -tests/fm-voice-relay.test.sh 32486 -tests/fm-wake-daemon-lifecycle-e2e.test.sh 7477 -tests/fm-wake-drain-open-decisions-cursor.test.sh 38506 -tests/fm-wake-drain-open-decisions.test.sh 6890 -tests/fm-wake-drain-outcome-backstop.test.sh 44076 -tests/fm-wake-drain-unread-status.test.sh 16169 -tests/fm-wake-queue.test.sh 85252 -tests/fm-watch-arm.test.sh 68479 -tests/fm-watch-checkpoint.test.sh 6076 -tests/fm-watch-recovery-loop.test.sh 58946 -tests/fm-watch-triage.test.sh 697969 -tests/fm-watcher-lock.test.sh 108940 +tests/fm-pr-check-security.test.sh 300675 +tests/fm-pr-reviewers.test.sh 157 +tests/fm-pr-state-live-e2e.test.sh 47 +tests/fm-pr-state.test.sh 525 +tests/fm-procevent-quota.test.sh 2459 +tests/fm-procevent-when.test.sh 25674 +tests/fm-procevent.test.sh 292297 +tests/fm-project-origin.test.sh 123 +tests/fm-public-followup.test.sh 381564 +tests/fm-quota-array-dispatch-live-e2e.test.sh 50 +tests/fm-quota-choose.test.sh 2860 +tests/fm-remote-backlog-handoff.test.sh 82063 +tests/fm-remote-doctor.test.sh 14460 +tests/fm-remote-entrypoint.test.sh 134 +tests/fm-remote-herdr-guard.test.sh 3140 +tests/fm-remote-job-orphan-reap.test.sh 2985 +tests/fm-remote-job.test.sh 81046 +tests/fm-remote-reply.test.sh 140887 +tests/fm-remote-secondmate-lifecycle-e2e.test.sh 345655 +tests/fm-remote-secondmate-parent-binding.test.sh 42294 +tests/fm-remote-secondmate-relaunch.test.sh 879 +tests/fm-remote-secondmate-trace-context.test.sh 74870 +tests/fm-remote-transport-lanes.test.sh 66089 +tests/fm-rovo-harness.test.sh 15691 +tests/fm-rovo-signals-live-e2e.test.sh 52 +tests/fm-secondmate-harness.test.sh 188187 +tests/fm-secondmate-lifecycle-e2e.test.sh 11268 +tests/fm-secondmate-liveness.test.sh 24564 +tests/fm-secondmate-reconcile.test.sh 100853 +tests/fm-secondmate-restart.test.sh 52591 +tests/fm-secondmate-safety.test.sh 69424 +tests/fm-secondmate-sync.test.sh 55501 +tests/fm-send-agy-confirm.test.sh 4440 +tests/fm-send-inbox-doorbell-live-e2e.test.sh 108 +tests/fm-send-inbox.test.sh 41713 +tests/fm-send-remote-delivery.test.sh 31964 +tests/fm-send-resolve-key.test.sh 47317 +tests/fm-send-secondmate-marker-herdr-e2e.test.sh 80 +tests/fm-send-secondmate-marker.test.sh 7574 +tests/fm-session-lock-ancestry.test.sh 18918 +tests/fm-session-start.test.sh 363574 +tests/fm-sessionstart-hook-live-e2e.test.sh 50 +tests/fm-sessionstart-instruction-refresh-live-e2e.test.sh 49 +tests/fm-sessionstart-nudge.test.sh 71802 +tests/fm-shared-captain-inheritance.test.sh 7991 +tests/fm-spawn-compact-adviser-disable-remote.test.sh 38561 +tests/fm-spawn-compact-adviser-disable.test.sh 21654 +tests/fm-spawn-dispatch-profile.test.sh 197548 +tests/fm-spawn-orca-worktree.test.sh 2400 +tests/fm-spawn-pool-base-freshen.test.sh 68652 +tests/fm-spawn-worktree-settle.test.sh 9309 +tests/fm-startup-memory-budget.test.sh 8086 +tests/fm-startup-network.test.sh 72106 +tests/fm-stat-shadowing.test.sh 75 +tests/fm-stow-cascade.test.sh 3058 +tests/fm-subagent-pretool-check.test.sh 998 +tests/fm-supervision-events.test.sh 673 +tests/fm-supervision-host-attended-live-e2e.test.sh 49 +tests/fm-supervision-host-live-e2e.test.sh 75 +tests/fm-supervision-host.test.sh 789123 +tests/fm-tangle-guard.test.sh 8501 +tests/fm-task-delivery.test.sh 32789 +tests/fm-task-inbox.test.sh 31965 +tests/fm-tasks-axi.test.sh 2293 +tests/fm-teardown-endpoint-safety.test.sh 40851 +tests/fm-teardown.test.sh 202132 +tests/fm-test-fixture-cleanup.test.sh 866 +tests/fm-test-fixtures.test.sh 1802 +tests/fm-test-isolation-proof.test.sh 2866 +tests/fm-timeout-lib.test.sh 10750 +tests/fm-tmux-agent-liveness.test.sh 3770 +tests/fm-tool-update-check.test.sh 14383 +tests/fm-trace-context-lib.test.sh 221 +tests/fm-trace-context-spawn.test.sh 57488 +tests/fm-turnend-foreign-owner-arm-fix.test.sh 5575 +tests/fm-turnend-guard.test.sh 34727 +tests/fm-update.test.sh 11894 +tests/fm-vendor-auth-probe.test.sh 43278 +tests/fm-voice-relay.test.sh 28917 +tests/fm-wake-daemon-lifecycle-e2e.test.sh 7345 +tests/fm-wake-drain-open-decisions-cursor.test.sh 47677 +tests/fm-wake-drain-open-decisions.test.sh 8781 +tests/fm-wake-drain-outcome-backstop.test.sh 46316 +tests/fm-wake-drain-unread-status.test.sh 24251 +tests/fm-wake-queue.test.sh 165906 +tests/fm-watch-arm.test.sh 113076 +tests/fm-watch-checkpoint.test.sh 11234 +tests/fm-watch-recovery-loop.test.sh 59092 +tests/fm-watch-triage.test.sh 1074843 +tests/fm-watcher-lock.test.sh 72022 +tests/fm-worker-account-live-e2e.test.sh 3179 +tests/fm-worker-account.test.sh 37445 EOF } @@ -877,6 +904,15 @@ portable_serial_unhinted() { rm -rf "$tmp" } +# Sum serial weights for paths on stdin, including the unmeasured default. +portable_serial_lane_weight() { + awk -v fallback="$PORTABLE_SERIAL_DEFAULT_WEIGHT_MS" ' + NR == FNR { if (NF) { hint[$1] = $2 }; next } + NF { total += ($1 in hint) ? hint[$1] : fallback } + END { printf "%d\n", total + 0 } + ' <(portable_serial_weight_hints) - +} + portable_parallel_weight_for() { local want=$1 ms ms=$(portable_parallel_weight_hints | awk -v want="$want" '$1 == want { print $2; exit }') @@ -1014,7 +1050,7 @@ select_lane() { } run_coverage_guard() { - local tmp missing extra a b shard unhinted serial_total + local tmp missing extra a b shard unhinted serial_total serial_ms serial_max_ms=0 local p1_ms p1_unhinted p2_ms p2_unhinted parallel_max_ms parallel_imbalance_ms local -a saved_scripts=() tmp=$(mktemp -d "${TMPDIR:-/tmp}/fm-test-coverage.XXXXXX") @@ -1060,6 +1096,8 @@ run_coverage_guard() { return 1 fi printf '%s\n' "${SCRIPTS[@]+"${SCRIPTS[@]}"}" >>"$tmp/serial_shards_raw" + serial_ms=$(printf '%s\n' "${SCRIPTS[@]+"${SCRIPTS[@]}"}" | portable_serial_lane_weight) + [ "$serial_ms" -le "$serial_max_ms" ] || serial_max_ms=$serial_ms shard=$((shard + 1)) done SCRIPTS=() @@ -1135,6 +1173,13 @@ run_coverage_guard() { return 1 fi + if [ "$serial_max_ms" -gt "$PORTABLE_SERIAL_MAX_WEIGHT_MS" ]; then + log "coverage guard: largest portable serial shard packs ${serial_max_ms}ms above the ${PORTABLE_SERIAL_MAX_WEIGHT_MS}ms target" + log "refresh CI hints and rebalance or add shards; do not raise the job timeout: docs/fm-test-portable-shards.md" + rm -rf "$tmp" + return 1 + fi + if [ -x "$ROOT/bin/fm-test-isolation-proof.sh" ]; then "$ROOT/bin/fm-test-isolation-proof.sh" --list | LC_ALL=C sort -u >"$tmp/proof_list" if ! cmp -s "$tmp/proven" "$tmp/proof_list"; then @@ -1154,7 +1199,7 @@ run_coverage_guard() { parallel_imbalance_ms=$((p1_ms - p2_ms)) [ "$parallel_imbalance_ms" -ge 0 ] || parallel_imbalance_ms=$((-parallel_imbalance_ms)) - printf 'FM_TEST_COVERAGE ok total=%s parallel=%s parallel_max_ms=%s parallel_imbalance_ms=%s parallel_unhinted=%s serial=%s serial_shards=%s serial_unhinted=%s herdr=%s\n' \ + printf 'FM_TEST_COVERAGE ok total=%s parallel=%s parallel_max_ms=%s parallel_imbalance_ms=%s parallel_unhinted=%s serial=%s serial_shards=%s serial_unhinted=%s serial_max_ms=%s serial_budget_ms=%s herdr=%s\n' \ "$(wc -l <"$tmp/all" | tr -d ' ')" \ "$(wc -l <"$tmp/shards_union" | tr -d ' ')" \ "$parallel_max_ms" \ @@ -1163,6 +1208,8 @@ run_coverage_guard() { "$(wc -l <"$tmp/serial" | tr -d ' ')" \ "$PORTABLE_SERIAL_SHARDS" \ "$unhinted" \ + "$serial_max_ms" \ + "$PORTABLE_SERIAL_MAX_WEIGHT_MS" \ "$(wc -l <"$tmp/herdr" | tr -d ' ')" rm -rf "$tmp" return 0 diff --git a/docs/architecture.md b/docs/architecture.md index 1617625cc69..99cad4719a2 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -131,7 +131,7 @@ The most recent recognized ci log marker wins, so checks-green monitoring report `bin/fm-crew-state.sh` owns the evidence guard that recognizes ended CI monitors after green checks, including cancelled runs and skipped rebase steps; a passed run alone never proves a forge merge. In the coarse runs-ledger fallback, which has no steps table and no ci log, a terminal failed record whose daemon an explicit `daemon status` probe proves down reports unknown as unverified instead: an instrument failure must never read as work failure. The same instrument rule covers the ledger-anchored continuation of a selected run whose head this copy cannot resolve: once the probe answers down, that still-executing record reports unknown as unverified, while a run parked at a gate keeps its gate and findings because an open decision stays open when the instrument dies, and a `needs-decision` or `blocked` event the crew observed first hand stays open with the unverified record named as the reason rather than superseded by it. -Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact idle permits fallback to the log's resolved current declaration - the newest decision the fold still holds open, otherwise a declared wait still standing after later resolved lines for other keys, otherwise the latest recognized event - when its verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. +Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact quota reports the stalled-on-a-provider-wall quota state, exact idle permits fallback to the log's resolved current declaration - the newest decision the fold still holds open, otherwise a declared wait still standing after later resolved lines for other keys, otherwise the latest recognized event - when its verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. The semantic branch reports working only on an exact busy verdict and names the source that produced it; an unknown verdict never becomes working, never permits the status-log fallback, and never becomes a silent idle. @@ -217,7 +217,7 @@ Pane existence, busy checks, composer checks, capture, and verified submit route The retries-exhausted queued-Enter decision is owned by `fm_composer_queued_enter_verdict` in `bin/fm-composer-lib.sh`; tmux and herdr provide only their backend-specific busy signals. Composer classification has one shared owner, `bin/fm-composer-lib.sh`: tmux, herdr, Zellij, Orca, and cmux contribute only a screen capture plus declarative styled, cursor, identity, and row capabilities, while the shared classifier owns every shape and the `empty`/`pending`/`pending-unproven`/`unknown` verdict. `fm-spawn.sh` also routes Kimi launch readiness through that classifier instead of carrying another shape copy. -The daemon injects only into an affirmatively `empty` composer, so every other or future verdict defers; positive container proof is required, and a blank unidentified row or bare dead-shell prompt cannot receive an escalation. +The daemon injects only into an affirmatively `empty` composer, so every other or future verdict defers; positive container proof or an identity-proven shape is required, and a blank unidentified row or bare dead-shell prompt cannot receive an escalation. The current operator boundary is in [Composer and injection safety](herdr-backend.md#composer-and-injection-safety). Unsupported supervisor backends refuse at daemon startup. Stalled escalation delivery writes `state/.subsuper-inject-wedged` and attempts a configured backend-independent active alert after `FM_MAX_DEFER_SECS` instead of silently deferring forever. @@ -234,16 +234,18 @@ Text for a worker to read and commands that drive a worker's process are separat ## Busy state is semantic, per adapter `bin/fm-busy-lib.sh` is the single owner of what "this worker is busy" means, and `bin/fm-busy-event.sh` is the only writer of the per-task records it reads. -Every classification returns a verdict of busy, idle, unknown, or dead together with the source that produced it, so a consumer or a diagnostic can never confuse semantic state with a fallback. +Every classification returns a verdict of busy, idle, unknown, dead, or quota together with the source that produced it, so a consumer or a diagnostic can never confuse semantic state with a fallback. Each converted adapter reports its own turn lifecycle through a machine-readable contract the vendor already exposes, rather than through rendered footer text: Pi and pi-signed through the Firstmate-owned extension's `agent_start` and `agent_settled` confirmed by `ctx.isIdle()`, omp through its extension's `agent_start` and `agent_end` without `willContinue`, OpenCode through its plugin's semantic `session.status`, Claude through owned `UserPromptSubmit`, `Stop`, `StopFailure`, and `SessionEnd` hooks, Muse through its session log, and Cursor through its conversation transcript. Kimi behind Pi inherits Pi's lifecycle. Codex and standalone Kimi classify unknown behind explicit probes until a semantic source is live-verified for them, and Grok, Rovo, and AGY each keep one clearly isolated rendered-tail busy fallback that can only ever classify their own task. -The one case where the contract reads rendered text for a converted adapter is the launch-prompt backstop (`fm_busy_launch_prompt_parked` in `bin/fm-busy-lib.sh`): when a record is still the untouched `fm-spawn` seed and the caller supplied a captured pane matching that harness's own recognized interactive launch prompt - a workspace-trust dialog, sign-in screen, or first-run menu - `fm_busy_classify` reports `unknown launch-prompt` instead of `busy fm-spawn`. +One case where the contract reads rendered text for a converted adapter is the launch-prompt backstop (`fm_busy_launch_prompt_parked` in `bin/fm-busy-lib.sh`): when a record is still the untouched `fm-spawn` seed and the caller supplied a captured pane matching that harness's own recognized interactive launch prompt - a workspace-trust dialog, sign-in screen, or first-run menu - `fm_busy_classify` reports `unknown launch-prompt` instead of `busy fm-spawn`. That keeps a launch that never began its brief from holding the busy-age exemption for the whole `FM_BUSY_TURN_MAX_SECS` bound and surfaces it through the ordinary not-provably-working path instead. A record any real hook event has advanced is never reclassified this way however its pane looks, no captured tail means the record's own state stands, and the general busy bound is unchanged. The per-harness signature table lives in `bin/fm-busy-lib.sh`'s header, and [runtime backend verification](verification/runtime-backends.md#launch-prompt-backstop-signatures) owns the live evidence. +The other case is the cross-harness provider quota wall (`fm_busy_quota_wall_verdict` in `bin/fm-busy-lib.sh`): it never invents a verdict from a rendered surface, only downgrades an already-busy semantic verdict to `quota` when the captured tail shows both a limit phrase and a retry or reset phrase, so a live harness stalled on a usage-limit retry modal is not read as working; [runtime backend verification](verification/runtime-backends.md#provider-quota-wall-classification) owns the evidence. + Missing, malformed, stale, untrusted, or unverified semantic state is unknown, never idle, and unknown is never promoted to busy either. Ordinary task-state consumers act only on an exact busy verdict, so an unreadable worker surfaces for a closer look instead of being absorbed as still-working or written off as finished. Endpoint death is the only process-level override and yields dead; child processes, CPU, process sleep state, and marker modification times are not state signals. diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index 89e84796271..acad64ddcd0 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -302,7 +302,7 @@ The same real-Pi reproduction then delivered the notification exactly once in a `tests/fm-calm-pi-extension.test.sh` compares wrapped and stock renderers and verifies all seven built-ins plus `fm_watch_arm_pi`; its rendered HTML export check accepts either omission or default-hidden hook rows for legacy synthetic messages while rejecting visible leakage. `tests/fm-pi-branch-extension.test.sh` verifies both `fm_branch_outcomes` and `fm_branch_processed` call headers against pre-0.99 and 0.99+ Pi stock rendering, plus Calm toggling, capability-probed all-line versus collapsed stock result output, exact expanded output, and export rendering for outcomes. Together they exercise redraw of already-rendered tool, thinking, current operational-user, and legacy synthetic rows, and cover every policy class. -It covers persisted preference restoration across every session-start reason and a real restart, proves the working-ship presentation and Calm-off stock `Working...` row through a delayed deterministic provider, asserts no Calm status row, verifies operational messages remain exact ordinary user-role session entries and complete exports, and drives genuine 100 by 44, 160 by 36, and 180 by 44 terminal fixtures. +It covers persisted preference restoration across every session-start reason and a real restart, proves the working-ship presentation and Calm-off stock `Working...` row through a delayed deterministic provider, asserts no Calm status row, verifies operational messages remain exact ordinary user-role session entries and complete exports, and drives genuine 100 by 44, 160 by 36, 180 by 44, and 180 by 120 terminal fixtures. A native deterministic `/skill:ahoy` turn produces thinking, tool-call, and tool-result blocks, asserts that the collapsed skill-to-final gap equals the two-row visible-only baseline, expands and re-collapses original thinking, restores Calm-off rendering, verifies persisted hidden history, and repeats the geometry assertion after restart with `terminal.clearOnShrink` explicitly off. The operational provider path covers Calm loaded on, loaded off, default preference, extension absent, exact watcher delivery, narrow bare-marker legacy input, persisted restart replay, a genuine captain prompt, and adjacent notifications coalesced into one intended processing turn. It asserts one persisted and rendered captain answer, exact user-role operational envelopes in order, no replacement custom messages, one processing result, zero operational transcript rows, and the two-row neighboring-assistant geometry for live, adjacent, and restart paths. diff --git a/docs/configuration.md b/docs/configuration.md index d22b42471e8..6275d09c2f1 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -494,6 +494,7 @@ Backend guides and other documents refer here instead of restating the resolutio `fm-teardown.sh ` takes a task id directly and validates the complete metadata-only endpoint identity before any runtime dispatch or cleanup mutation. Missing, empty, duplicate, malformed, backend-inconsistent, or task-mismatched endpoint records are preserved and refused. +A windowless record carrying an explicit `endpoint_cleared=` stamp is accepted as already-closed agent-less evidence with no flag and no `--force`, which is how a lane whose pane or workspace was closed by hand can still be retired; the unlanded-work refusal is unchanged, and [`bin/fm-teardown.sh`](../bin/fm-teardown.sh)'s header owns the stamp's shape. Legacy tmux metadata remains cleanup-compatible when its exact window name is `fm-`; opaque non-tmux endpoints require their recorded `endpoint_task_id=` binding. @@ -746,7 +747,7 @@ A local standalone-clone home cannot receive a primary-local commit through that ## Harness support -claude, codex, opencode, pi, pi-signed, grok, kimi, cursor, and omp are empirically verified for crewmate and secondmate launches; gemini and cline are verified for crewmate and scout launches only, and [README requirements](../README.md#requirements) own the set supported for the primary session. +claude, codex, opencode, pi, pi-signed, grok, kimi, cursor, and omp are empirically verified for crewmate and secondmate launches; the adapters below that are verified for crewmate and scout launches only are each refused for a secondmate, and [README requirements](../README.md#requirements) own the set supported for the primary session. ### Harness restrictions and credentials `fm-spawn.sh` refuses kimi on cmux and Orca at preflight, because answering Kimi's folder-trust dialog needs a verified viewport-only capture those backends lack; [its adapter reference](../.agents/skills/harness-adapters/references/harness/kimi.md#readiness-gated-start) owns the trust-dialog handling. @@ -768,7 +769,7 @@ cline is likewise verified for crewmate and scout launches ONLY, refused for a s openhands is likewise verified for crewmate and scout launches ONLY, refused for a secondmate for the same reason - no hook surface and no primary supervision protocol; [`docs/verification/openhands.md`](verification/openhands.md) owns that evidence, including the per-task HOME required because the SDK profile store is hardcoded under `~/.openhands/profiles`. openhands also needs `LLM_API_KEY` before spawning, taken from the environment or from the optional gitignored `config/openhands-llm.env`, and `LLM_MODEL` from `--model` (a LiteLLM id such as `fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash`). -Its private worker config disables Claude Code imports (including the captain's hooks) and, unless the home sets `config/keep-ai-trailers` (see "Commit attribution"), Devin commit attribution without editing user or project config; [`fm-devin-config.sh`](../bin/fm-devin-config.sh) owns these enforced settings and [Devin verification](verification/devin.md) owns the live evidence and observed model availability. +Devin's private worker config disables Claude Code imports (including the captain's hooks) and, unless the home sets `config/keep-ai-trailers` (see "Commit attribution"), Devin commit attribution without editing user or project config; [`fm-devin-config.sh`](../bin/fm-devin-config.sh) owns these enforced settings and [Devin verification](verification/devin.md) owns the live evidence and observed model availability. ### Verification and primary supervision New harnesses get verified through a supervised trial task before joining the set. @@ -1015,7 +1016,7 @@ The optional local, gitignored `config/keep-ai-trailers` presence flag opts this With the flag absent, every Claude launch's inline `--settings` JSON carries `"attribution":{"commit":"","pr":"","sessionUrl":false}`, every Devin worker config sets `"attribution": false`, and every fleet launch receives a pane-scoped `GIT_CONFIG` `core.hooksPath` pointing at `state/.git-hooks`, where git's `commit-msg` hook strips known AI trailers even when a runtime injects them after the typed message. When the flag is present, Claude launches omit those attribution-off settings, Devin worker configs keep the user config's `attribution` setting (Devin's default is on), and fleet launches do not install or select the strip hooks, so Git uses the repository's configured hooks directly. `bin/fm-git-strip-ai-trailers.sh` owns the identities, the install, and chaining the hooks of whichever repository git is running in, including when `git -c core.hooksPath` supplies the pane's hook override, so a project hook such as husky still runs when stripping is enabled. -If the wrapper cannot resolve that repository's hooks directory, the git operation fails rather than silently skipping a project hook such as a pre-push guard. +A repository whose config sets `core.hooksPath` to the empty string runs no project hook, as in plain git; if the wrapper otherwise cannot resolve that repository's hooks directory, the git operation fails rather than silently skipping a project hook such as a pre-push guard. When stripping is enabled, the hooks directory is read-only, so a hook manager run inside a fleet pane (lefthook's npm postinstall, `pre-commit install`) fails instead of displacing the strip; install a project's hooks from outside the pane, where the wrappers chain them. The flag is a home-wide attribution choice, so it is inherited into secondmate homes under the [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md) inherited-local-material contract and a secondmate's own workers keep AI trailers too. Per-machine Cursor `cli-config.json` attribution-off is not this contract: it does not travel with Firstmate, defaults back to on when unset, and only feeds the CLI's request to the server, so it suppresses the trailer rather than preventing it. @@ -2321,6 +2322,7 @@ FM_PROC_ROOT_OVERRIDE= # alternate /proc root for Linux process-identity reads FM_BACKEND= # optional runtime backend override for new spawns; tmux/herdr/zellij/orca/cmux support ship/scout spawns, codex-app is not accepted FM_TRACE_CONTEXT= # optional trace-context override; see "Trace context propagation" FM_TASK_ID= # internal task-worker marker fm-spawn.sh exports into ship and scout panes, never set by hand; bin/fm-test-run.sh refuses to execute in the repository primary checkout while it is set +FM_TASK_INBOX= # internal: absolute path of the task's steering inbox (state/.inbox) that fm-spawn.sh exports into every ship, scout, and secondmate launch, never set by hand; the steering doorbell names "$FM_TASK_INBOX" HERDR_SESSION=default # herdr-only: named session for normal backend ops; not enough for destructive cleanup (docs/herdr-backend.md) FM_BACKEND_HERDR_SUBMIT_POLLS=6 # herdr-only: agent-state samples spread across each Enter attempt's budget when confirming a submit (docs/herdr-backend.md "Current transport behavior") FM_BACKEND_HERDR_SUBMIT_MIN_SLEEP=0.6 # herdr-only: minimum per-Enter confirmation budget before polling agent-state after an idle baseline diff --git a/docs/fm-test-portable-shards.md b/docs/fm-test-portable-shards.md index 859ab9da099..4e760aa7c95 100644 --- a/docs/fm-test-portable-shards.md +++ b/docs/fm-test-portable-shards.md @@ -9,27 +9,27 @@ Balance hints come from serial runs of the real lanes on `ubuntu-latest`. The concurrent isolation proof in [fm-test-isolation-proof.md](fm-test-isolation-proof.md) establishes concurrency safety, not serial CI duration. Local timings are not interchangeable with CI timings: platform and machine load can affect each script differently and change their relative weights. -The retained hints are the slowest completed value each script reached across six CI runs on 2026-09-10: [34459949083](https://github.com/kunchenguid/firstmate/actions/runs/34459949083), [34460760299](https://github.com/kunchenguid/firstmate/actions/runs/34460760299), [34462530836](https://github.com/kunchenguid/firstmate/actions/runs/34462530836), [34462758357](https://github.com/kunchenguid/firstmate/actions/runs/34462758357), [34466966385](https://github.com/kunchenguid/firstmate/actions/runs/34466966385), and [34470382458](https://github.com/kunchenguid/firstmate/actions/runs/34470382458). -Shard 2 completed in all six, so its scripts come from the uploaded `fm-test-timing-portable-parallel-2` artifacts. -Shard 1 was cancelled at its job cap in five of the six, so its scripts come from the `FM_TEST_END duration_ms=` markers in each cancelled job's log, which record every script that finished before the cancellation, plus the one complete `fm-test-timing-portable-parallel-1` artifact from run 34462758357. +Both hint tables were refreshed on 2026-09-30 from five Ubuntu CI runs: [36583881812](https://github.com/kunchenguid/firstmate/actions/runs/36583881812), [36658498535](https://github.com/kunchenguid/firstmate/actions/runs/36658498535), [36663947738](https://github.com/kunchenguid/firstmate/actions/runs/36663947738), [36664663190](https://github.com/kunchenguid/firstmate/actions/runs/36664663190), and [36669175457](https://github.com/kunchenguid/firstmate/actions/runs/36669175457). +Use the slowest successful `duration_ms` per script across their uploaded portable timing artifacts and completed `FM_TEST_END` log markers, with the two version/platform exceptions below. +All artifact records were cross-checked against the corresponding job's markers. +Those runs are upstream's, and their records cover all 24 parallel members and the 201 serial members the lane held upstream; an existing live-capability skip is a portable-runner measurement, not a timing claim for the unavailable live integration. +This fork's serial lane carries additional fork-only members that no upstream run measured, so each of them packs on the `PORTABLE_SERIAL_DEFAULT_WEIGHT_MS` default until the fork refreshes its own hints from its own green CI runs; read the current lane size and unmeasured share from `bin/fm-test-run.sh --check-coverage` rather than from a count copied here. Observed maxima provide conservative packing weights, not an upper bound on future durations. -The measurements cover all 24 candidates, with six samples per script except: +Two serial-5 jobs were cancelled at their 30-minute cap and uploaded no artifact. +Their completed log markers supplement the complete runs, but a cancelled job's wall time is only a lower bound and its unfinished or never-started scripts have no completed sample. +A failed script's duration is excluded even when its lane uploaded an artifact. +In particular, run 36664663190's serial 5 finished in 22m15s with an assertion failure, not a timeout; treating that as a healthy whole-lane sample would hide the failure. +Collect successful per-script measurements for every member before calculating a split. -| Samples | Scripts | -|---:|---| -| 4 | `tests/fm-lint.test.sh` | -| 3 | `tests/fm-pi-primary-types.test.sh`, `tests/fm-review-diff.test.sh` | -| 1 | `tests/fm-brief.test.sh`, `tests/fm-transition-lib.test.sh` | - -The two scripts with one sample are the tail of shard 1 that only the complete run reached. -Collect completed per-script measurements for every member before calculating a split. -A cancelled lane's elapsed duration is only a lower bound; its unfinished scripts have no completed duration for that invocation. -The complete historical run supplies tail-script hints, not a completion time for any later cancelled invocation or for the rebalanced jobs. +`tests/fm-supervision-host.test.sh` uses 789123 ms from run 36669175457, after the merged [host runtime fix](https://github.com/kunchenguid/firstmate/pull/6179), rather than its pre-fix maximum of 1065298 ms. +That post-fix value has only one sample in this baseline, so further green runs must establish its variance. +The native-Windows-only `tests/fm-pi-windows-shell-invocation.test.sh` retains its separate 5121 ms measurement from 2026-09-06T21:02Z instead of a portable capability skip. +The session-start hint retains its pre-optimization maximum until CI measures the shorter fixture-only home-summary bound; do not discount a local speedup from CI packing weights. ## Parallel lanes -The two parallel lanes use longest-processing-time assignment over those hints. +The two parallel lanes use longest-processing-time assignment over those hints, with the Pi typecheck pinned to the job that installs its prerequisite. [`bin/fm-test-run.sh`](../bin/fm-test-run.sh) holds the duration values in `portable_parallel_weight_hints` and the ordered memberships and lane-specific prerequisite constraints beside `list_portable_parallel_1` and `list_portable_parallel_2`. Read the derived packing estimates with that runner's `--check-coverage`; its header and `--help` own the output fields and the selection-specific `--list-scheduled` weight rules. The largest individual hint sets a lower bound on the estimated duration of any split, regardless of how evenly the remaining work is assigned. @@ -57,21 +57,21 @@ Each shard is still strictly serial in itself, and separate runners mean no two `.github/workflows/ci.yml` derives the same `n` from `strategy.job-total` rather than a literal, so changing the shard count in either file without the other fails the lane loudly instead of leaving part of the required suite unrun. Assignment is longest-processing-time bin packing over per-script duration hints embedded in `bin/fm-test-run.sh`. -The serial hints were refreshed from successful per-script records in the `fm-test-timing-portable-serial-*` artifacts of the complete green [run 35279383618](https://github.com/kunchenguid/firstmate/actions/runs/35279383618) and the available completed shards of [run 35282466441](https://github.com/kunchenguid/firstmate/actions/runs/35282466441) on 2026-09-17. -Together these cover all 176 serial scripts at refresh time; retain the slower successful sample where both exist. -The native-Windows-only `tests/fm-pi-windows-shell-invocation.test.sh` retains its separate 5121 ms measurement from 2026-09-06T21:02Z instead of a portable capability skip. -An unfinished or failed invocation is not a healthy duration sample. +[Verification inputs](#verification-inputs) owns the measurement provenance and exceptions. A script with no hint gets the conservative `PORTABLE_SERIAL_DEFAULT_WEIGHT_MS` default. Hints only affect balance: the coverage guard keeps the partition complete and disjoint whatever they say, so a stale hint costs a slower shard rather than lost coverage. Balance is still worth keeping current, because enough unmeasured scripts let one shard carry more than twice another shard's real work and reach the job cap while another runner sits idle. -That is not hypothetical: by 2026-09-01 the lane had grown from 116 to 139 scripts and from ~42 to ~63 minutes, 17 scripts were still unmeasured, and several hints were low by 2-5x, so shard 3 of 4 ran 17-20 minutes against its 20-minute cap while shard 1 ran 11.5 minutes and run [33574154856](https://github.com/kunchenguid/firstmate/actions/runs/33574154856) timed out seconds after a passing test. -`bin/fm-test-run.sh --check-coverage` now reports the unmeasured share as `serial_unhinted=` and refuses past `PORTABLE_SERIAL_MAX_UNHINTED_PERCENT`, so hint drift fails the coverage guard instead of silently pushing one shard into its job cap. -Refresh the hints whenever the serial lane gains scripts, rather than waiting for that bound to trip. +`bin/fm-test-run.sh --check-coverage` reports the unmeasured share as `serial_unhinted=` and refuses past `PORTABLE_SERIAL_MAX_UNHINTED_PERCENT`. +That catches missing hints, not stale existing hints: the host suite still had a 41512 ms hint after growing to over 1000 seconds in CI, so the old split placed it beside another 12 minutes of work while passing the guard. +Refresh the hints whenever a serial member grows materially or the lane gains scripts, rather than waiting for missing-hint coverage to trip. `bin/fm-test-run.sh` owns the per-shard packing, so its `--check-coverage` output is the current account of lane size and coverage rather than a copied inventory. -Nine serial runners pack the refreshed measurements into a longest modeled script sum of 697969 ms (11m38s), with other shards near 10m36s. -The longest script, `tests/fm-watch-triage.test.sh`, legitimately occupies one whole shard and is the indivisible floor for this layout. -This is a packing estimate, not measured new-workflow execution or an end-to-end latency guarantee. +Its header and `--help` own the modeled-budget check and output fields; read the current estimates from `--check-coverage` instead of retaining copied lane sums here. +[`tests/fm-test-run.test.sh`](../tests/fm-test-run.test.sh), in `test_portable_serial_packing_budget_boundary`, verifies acceptance exactly at the budget and refusal one millisecond above it through the executable runner. +The longest script, `tests/fm-watch-triage.test.sh`, is the indivisible floor for this layout. +The estimates use per-file maxima from different runs, not measured rebalanced jobs or an end-to-end latency guarantee. +The baseline watch-triage samples range from 944375 to 1074843 ms, while each observed completed portable job adds at most 30 seconds beyond its summed scripts in these runs. +Even so, maxima from five runs do not establish a P95 or guarantee future headroom. Job timeouts remain hang tripwires under the policy in [Timeouts](#timeouts) below; they are not the desired healthy duration. `tests/fm-ci-workflow.test.sh` compares the parsed CI matrix to the executable runner lanes, and the runner rejects parallel `--jobs` on a serial lane even when that shard has only one member. @@ -79,14 +79,15 @@ Refresh the CI-derived hints by downloading the per-shard timing artifacts from ```sh for run in ; do - gh run download "$run" -R kunchenguid/firstmate --pattern 'fm-test-timing-portable-serial-*' -D "/tmp/fm-serial/$run" + gh-axi run download "$run" -R /firstmate --dir "/tmp/fm-serial/$run" done -jq -r '.scripts[] | select(.exit == 0) | [.path, .duration_ms] | @tsv' /tmp/fm-serial/*/*/*.json \ +jq -r '.scripts[] | select(.exit == 0) | [.path, .duration_ms] | @tsv' /tmp/fm-serial/*/fm-test-timing-portable-serial-*/*.json \ | awk -F'\t' '$2 > m[$1] { m[$1] = $2 } END { for (p in m) print p, m[p] }' \ | LC_ALL=C sort bin/fm-test-run.sh --check-coverage ``` +Name the repository whose lane you are refreshing: an upstream run never executes a fork-only member, so it cannot supply that member's hint. A timed-out shard may upload no artifact, so include a complete green run or the slowest scripts go unmeasured in exactly the shard that needs them most. Completed shards from a partial run can supplement that complete baseline, but never treat missing tail scripts or the timeout duration as successful samples. Measure native-Windows-only scripts through the focused Git Bash runner and retain that `duration_ms` separately, because the portable CI shards skip them. @@ -96,7 +97,7 @@ Measure native-Windows-only scripts through the focused Git Bash runner and reta `bin/fm-test-run.sh --check-coverage` verifies that both parallel lanes partition the proven-isolated set. It also verifies that the parallel lanes, portable serial lane, and real-Herdr family are disjoint and cover every `tests/*.test.sh` script. It separately verifies that the portable serial CI shards are non-empty, disjoint, and together equal the portable serial lane. -It reports the unmeasured serial share as `serial_unhinted=` and refuses when that share exceeds `PORTABLE_SERIAL_MAX_UNHINTED_PERCENT`, so the shards stay balanced on evidence rather than on the default weight. +Its hint-coverage and modeled-budget checks are described in [Portable serial CI shards](#portable-serial-ci-shards); neither replaces inspection of actual CI timing artifacts. ## Timing artifacts @@ -112,8 +113,9 @@ Its `--list-files` interface exposes partition membership; `tests/fm-lint.test.s The workflow uploads each partition's quiet telemetry plus its per-root lifecycle sidecar to distinguish analysis cost, memory use, and host contention. No fast mode, path skips, reduced checks, or paid runner provisioning is part of this layout. -The performance objective is a complete green run under fifteen minutes including start delay: roughly twelve minutes of longest-path execution, at most two minutes of runner delay, and less than one minute of other overhead. -The candidate uses fourteen long-lived Linux jobs (nine serial, two parallel, Herdr, two lint), plus short checks and macOS; insufficient shared account capacity can erase the packing gain. +The longer-term performance objective remains a complete green run under fifteen minutes including start delay, but the current watch-triage floor alone exceeds that objective. +The immediate packing target is the runner's modeled script budget, not a claim that more shards alone can make an indivisible script faster. +The layout uses fourteen long-lived Linux jobs (nine serial, two parallel, Herdr, two lint), plus short checks and macOS; insufficient shared account capacity can erase the packing gain. Compare complete before/after runs, preserve cancelled and partial-run evidence, and measure a representative normal-run sample before claiming a P95 improvement. The workflow retains per-PR supersession without cancelling main pushes or changing the compliance workflow's event semantics. @@ -126,14 +128,15 @@ The workflow retains per-PR supersession without cancelling main pushes or chang CI job timeouts follow one three-tier policy, so the workflow reads as a policy rather than as a collection of per-job numbers. Every tier is a hang tripwire with headroom above the healthy duration, never a packing estimate or a runtime target. -A lane that reaches its tier bound is wedged, not slow, so change the policy here rather than treating the bound as a way to fit a slower lane. +A lane that reaches its tier bound needs investigation and a distribution or runtime fix before the tier changes again. +Raising a bound is a deliberate owner decision that must cite observed evidence of lost headroom - a job killed at the bound with its assertions still passing - not a way to fit the same work into a slower lane. | Tier | Jobs | Bound | Rationale | |---|---|---|---| | Fast | coverage guard, repo invariants, timing aggregate | 5 minutes | Seconds-long local work, so the tripwire only catches a hung runner. | -| Normal | lint partitions, portable parallel shards, portable serial shards, macOS stock Bash | 30 minutes, one value shared by every job in the tier | One shared hang tripwire keeps every ordinary test and lint lane on the same policy instead of allowing per-lane packing estimates or one-off caps to set the bound. | +| Normal | lint partitions, portable parallel shards, portable serial shards, macOS stock Bash | 60 minutes, one value shared by every job in the tier | One shared hang tripwire keeps every ordinary test and lint lane on the same policy instead of allowing per-lane packing estimates or one-off caps to set the bound. | | Heavy | Herdr | family-run step 20 minutes under a 75-minute job-level last-resort backstop | Healthy runs finish in about 7-10 minutes, so the step tripwire fails a wedged suite while the `always()` cleanup and timing upload still run, and the job cap only catches a hang outside that step. | [`.github/workflows/ci.yml`](../.github/workflows/ci.yml) holds the executable values and names each job's tier beside its `timeout-minutes`. -[`tests/fm-ci-workflow.test.sh`](../tests/fm-ci-workflow.test.sh) holds the policy against the parsed workflow: every job belongs to exactly one tier, the workflow carries exactly three distinct job-level values, the fast tier stays within 5-10 minutes, the normal jobs share one 30-minute budget, and the Herdr family-run step is the 20-minute tripwire below its job backstop with an `always()` teardown after it. +[`tests/fm-ci-workflow.test.sh`](../tests/fm-ci-workflow.test.sh) holds the policy against the parsed workflow: every job belongs to exactly one tier, the workflow carries exactly three distinct job-level values, the fast tier stays within 5-10 minutes, the normal jobs share one 60-minute budget, and the Herdr family-run step is the 20-minute tripwire below its job backstop with an `always()` teardown after it. A passing coverage guard does not establish a healthy job duration; refresh the healthy figures above from the lanes' uploaded timing artifacts. diff --git a/docs/herdr-backend.md b/docs/herdr-backend.md index 1abcb62987c..b79045e10cd 100644 --- a/docs/herdr-backend.md +++ b/docs/herdr-backend.md @@ -633,6 +633,7 @@ It hands the visible pane's ANSI viewport plus Herdr's capability facts to the f - Bare agent-glyph rows, including muse's `⟩`, which the adapter's retired local pattern silently omitted. - opencode's left bar. - The Pi separator region this adapter pioneered, admitted only when native `agent get` identity is exactly Pi and state is idle or done. +- The agy composer row - a bare `>` above a full-width rule - admitted only when native `agent get` identity is exactly agy. ### Pi composer states diff --git a/docs/tmux-backend.md b/docs/tmux-backend.md index 3f1b056f13a..4da10bfef85 100644 --- a/docs/tmux-backend.md +++ b/docs/tmux-backend.md @@ -48,7 +48,7 @@ Verify setup by spawning a small task and confirming its `fm-` window appear A target-existence check proves only that the pane exists. The deeper tmux agent-liveness probe first verifies exact window membership, then reads process names to distinguish a running harness from a bare idle shell. -It classifies recognized Claude, Codex, OpenCode, Pi, pi-signed, Grok, Kimi, Cursor, Muse, Rovo, AGY, and cline process identities as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. +It classifies recognized Claude, Codex, OpenCode, Pi, pi-signed, Grok, Kimi, Cursor, Muse, Rovo, AGY, cline, and openhands process identities as `alive`, common shells as `dead`, an authoritatively absent window as `missing`, unreadable state as `unreadable`, and every other process as `ambiguous`. The process-name vocabulary behind those verdicts is owned by `bin/fm-agent-process-lib.sh` and shared with the Herdr adapter, which proves a registered agent against the same names ([herdr-backend.md](herdr-backend.md) "Restart and liveness behavior"). Only `dead` and `missing` authorize recovery because a false dead result could launch a duplicate agent. @@ -64,6 +64,7 @@ Muse is likewise anchored to the exact `muse` launcher identity or the installed omp is anchored to the exact `omp` identity for the same reason, so `ompd` and `comp` remain ambiguous. AGY and Devin are anchored to the exact `agy` and `devin` identities for the same reason, so unrelated names containing either fragment remain ambiguous. cline is anchored to the exact `.cline` native-binary identity, with a node-wrapper fallback on the anchored path fragments `/bin/cline` and `@cline/cli`, so a script merely containing "cline" is not accepted. +openhands is anchored to the exact `openhands` process name for the same reason, so a path or argument merely containing `.openhands` remains ambiguous. Cursor is identified from its exact `cursor-agent` identity or versioned install tree in the foreground process path or structured argv[0]; a bare `node` or unrelated `agent` remains ambiguous. The CI-enforced portable regression and opt-in real-harness drift guard follow the split owned by `.agents/skills/firstmate-coding-guidelines/SKILL.md`. @@ -76,11 +77,11 @@ The tmux reader is a thin adapter over the fleet-wide classifier in `bin/fm-comp Real text in an identified shape is pending, while only positively proven emptiness reads empty. A blank or otherwise unidentified cursor row is `unknown` and every consumer defers, except that a foreground process proven to be Cursor is re-read cursorlessly because Cursor parks its terminal cursor below its footer. That identity-gated exception preserves the strict container-proof rule for every other pane, so a modal dialog, a dead shell between stale rules, or a mid-redraw pane is never an injection target. -The shared classifier accepts a shell glyph as an empty agent composer only inside a bordered container. +The shared classifier accepts a shell glyph as an empty agent composer only inside a bordered container, or as agy's bare `>` composer row, which it proves empty only with a live agy identity. A bare shell prompt is `unknown`, so away-mode escalation is never injected into a dead shell. -Busy state is not read from rendered text on this backend. -A task's busy, idle, unknown, or dead verdict comes from the semantic busy-state contract owned by `bin/fm-busy-lib.sh`; [architecture](architecture.md#busy-state-is-semantic-per-adapter) owns its boundaries. +Busy state is not read from rendered text on this backend, apart from the shared contract's cross-harness provider quota-wall downgrade. +A task's busy, idle, unknown, dead, or quota verdict comes from the semantic busy-state contract owned by `bin/fm-busy-lib.sh`; [architecture](architecture.md#busy-state-is-semantic-per-adapter) owns its boundaries. The isolated rendered-tail busy fallbacks that remain are harness-scoped, so one adapter's output can never classify another's task. The submit acknowledgement and away-mode supervisor-pane busy guard below still consult rendered output, but only to decide whether input can be delivered, never to decide recorded task state. The supervisor guard selects only the detected primary harness's signature rather than a global union of vendor patterns. diff --git a/docs/verification/runtime-backends.md b/docs/verification/runtime-backends.md index 0cefa442d2f..2f1b33af792 100644 --- a/docs/verification/runtime-backends.md +++ b/docs/verification/runtime-backends.md @@ -851,6 +851,24 @@ The current pending-composer ring contract is owned by `bin/fm-task-inbox-lib.sh Kimi was not installed on the verification machine; its receive path is the same one-line-plus-shell contract, and the portable ladder and enqueue regressions in `tests/fm-task-inbox.test.sh` and `tests/fm-send-inbox.test.sh` cover every harness-independent half. This guard is the refresh command after any harness upgrade; it spends a small number of real tokens per installed harness, reports an absent harness explicitly, and refuses a run that verified nothing. +The doorbell no longer prints the inbox's absolute path, so its length no longer grows with the home's depth. +It names the inbox as `"$FM_TASK_INBOX"`, which `bin/fm-spawn.sh` exports into every launch as the absolute `state/.inbox` path, followed by the short `.inbox` name; the brief's full path remains the fallback for a worker launched without that export. +The guard now launches each worker with `FM_TASK_INBOX` exported and no brief, so the worker must resolve the inbox from the doorbell and its environment alone. +It is the refresh command for that shape, which has not yet been recorded live here. +The run below, on 2026-09-30 on tmux 3.6, Linux (WSL2), with the same command, covered the earlier brief-primed shape, whose doorbell named only the short `.inbox` name and whose guard gave each worker the brief's steering-inbox sentence before the steer: + +```text +ok - claude (2.1.285 (Claude Code)): the doorbell reached a real worker, which acted and acked with the mv +ok - codex (codex-cli 0.157.0): the doorbell reached a real worker, which acted and acked with the mv +ok - opencode (1.18.33): the doorbell reached a real worker, which acted and acked with the mv +# harness absent, not verified here: grok +# harness absent, not verified here: kimi +# harness absent, not verified here: muse +``` + +OpenCode needed `FM_SEND_INBOX_LIVE_TIMEOUT=560` because its configured model was still mid-turn at the default 240 seconds. +Pi 0.87.1 was installed but not verified: its configured model returned an account error (`The 'gpt-5.6-sol' model is not supported when using Codex with a ChatGPT account`) before it read the inbox. + ## Gemini The Gemini crewmate adapter was verified on 2026-09-04 with gemini-cli 0.58.0 on Linux, Node v24.20.0, tmux 3.4. diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index e73c0948c98..25e7759a145 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -20,6 +20,15 @@ PI_OPERATIONAL_INPUT="$ROOT/.pi/extensions/lib/fm-operational-input.ts" PI_PACKAGE_DIR=${FM_PI_PACKAGE_DIR:-"$(npm root -g 2>/dev/null)/@earendil-works/pi-coding-agent"} TMUX_SOCKET="fm-calm-$$" TMUX_SESSION="fm-calm-e2e" +# Restored E2E transcripts can exceed 600 lines, so any capture that must see +# the earliest restored rows uses full scrollback; a fixed window silently +# drops the top of the transcript as later turns render and has produced a +# deterministic CI failure (missing CALM_E2E_OUTPUT) with the same suite green +# locally. +capture_full() { + local file=$1 + tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S - >"$file" 2>/dev/null || true +} # Verified against Pi 0.81.1 and 0.82.0 (docs/calm-mode-feasibility.md). This is # known-good evidence, not a support ceiling: the fixtures below run against whatever # Pi is actually installed, and record_pi_version_evidence never rejects a newer @@ -42,10 +51,10 @@ trap cleanup EXIT wait_for_text() { local file=$1 text=$2 i=0 while [ "$i" -lt 120 ]; do - # Include recent scrollback: expanding a long restored transcript can move - # the asserted tool output above the current viewport while the footer and - # editor remain visible. - tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S -600 >"$file" 2>/dev/null || true + # Include the full scrollback: a long restored transcript keeps growing + # above the viewport, so a fixed window can drop the earliest asserted + # rows (the restored tool output) off the top as later turns render. + capture_full "$file" grep -Fq "$text" "$file" 2>/dev/null && return 0 sleep 0.05 i=$((i + 1)) @@ -2618,7 +2627,7 @@ TS elif [ "$calm_state" = on ]; then # Nothing appears to wait for, so give Pi's listing a moment to repaint. sleep 1 - tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S -600 >"$TMP_ROOT/queued-escape-pane" + capture_full "$TMP_ROOT/queued-escape-pane" else wait_for_text "$TMP_ROOT/queued-escape-pane" "Follow-up:" \ || fail "Pi queued-row $label case never listed the queued notification" @@ -4172,12 +4181,12 @@ TS cat >"$session_file" <"$hidden_snapshot" + capture_full "$hidden_snapshot" # Wait for the redraw this block actually asserts: the collapsed-thinking adapter # (unconditional, unaffected by the built-in tool gate below) hides, and the # retained genuine rows are back on screen. Built-in tool rows from before this @@ -4476,7 +4493,7 @@ JS # confirmation, so it must not overwrite it: the captain has to keep seeing where # their export landed. The export-data assertions above take seconds of real time, # so this snapshot is taken well after that repaint has settled rather than racing it. - tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S -600 >"$export_settled_snapshot" + capture_full "$export_settled_snapshot" assert_contains "$(cat "$export_settled_snapshot")" "Session exported to: $export_file" \ "Calm's post-export repaint overwrote Pi's export confirmation" assert_not_contains "$(cat "$export_settled_snapshot")" "fm_watch_arm_pi" \ @@ -4746,7 +4763,7 @@ JS tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" Escape active_screen_wait=0 while [ "$active_screen_wait" -lt 200 ]; do - tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S -600 >"$boat_cleared_snapshot" + capture_full "$boat_cleared_snapshot" if ! grep -Fq '╲▁▁▁╱' "$boat_cleared_snapshot" && [ "$(grep -Fc 'Operation aborted' "$boat_cleared_snapshot" || true)" -ge 1 ]; then break @@ -4797,7 +4814,7 @@ JS tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" Escape active_screen_wait=0 while [ "$active_screen_wait" -lt 200 ]; do - tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S -600 >"$boat_cleared_snapshot" + capture_full "$boat_cleared_snapshot" if ! grep -Fq '╲▁▁▁╱' "$boat_cleared_snapshot" && [ "$(grep -Fc 'Operation aborted' "$boat_cleared_snapshot" || true)" -ge 2 ]; then break diff --git a/tests/fm-ci-workflow.test.sh b/tests/fm-ci-workflow.test.sh index 1fdcbd67821..c4ff2206564 100755 --- a/tests/fm-ci-workflow.test.sh +++ b/tests/fm-ci-workflow.test.sh @@ -173,7 +173,7 @@ test_fast_tier_shares_one_short_tripwire() { pass "fast tier jobs share one $fast minute tripwire" } -# Normal tier: every test or lint lane shares ONE fixed 30-minute budget, +# Normal tier: every test or lint lane shares ONE fixed 60-minute budget, # above the fast tier. That budget is a hang tripwire, not a packing estimate. test_normal_tier_shares_one_budget() { local fast normal @@ -183,8 +183,8 @@ test_normal_tier_shares_one_budget() { normal=$(tier_timeout normal $NORMAL_TIER_JOBS) || exit 1 [ "$normal" -gt "$fast" ] \ || fail "normal tier ($normal) must exceed the fast tier ($fast)" - [ "$normal" = 30 ] \ - || fail "normal tier must be the single 30-minute shared budget, got $normal" + [ "$normal" = 60 ] \ + || fail "normal tier must be the single 60-minute shared budget, got $normal" pass "normal tier jobs share one $normal minute budget" } diff --git a/tests/fm-claude-trust.test.sh b/tests/fm-claude-trust.test.sh index cafcb327dfd..70a807c04fa 100755 --- a/tests/fm-claude-trust.test.sh +++ b/tests/fm-claude-trust.test.sh @@ -626,11 +626,14 @@ test_refused_spawn_leaves_no_task_state() { } # Resolve the final prompt argument using the same shell argument splitting the -# pane sees after the two leading export statements. +# pane sees after the leading export statements. claude_launch_doorbell() { # - local command=${1#*; } + local command=$1 + while [[ "$command" == export\ *\;* ]]; do + command=${command#*; } + done ( - eval "set -- ${command#*; }" + eval "set -- $command" printf '%s' "${!#}" ) } diff --git a/tests/fm-git-strip-ai-trailers.test.sh b/tests/fm-git-strip-ai-trailers.test.sh index c12124194ba..b0ccd0fe8ed 100644 --- a/tests/fm-git-strip-ai-trailers.test.sh +++ b/tests/fm-git-strip-ai-trailers.test.sh @@ -234,6 +234,95 @@ test_pane_hookspath_does_not_reroute_another_repository() { pass "a pane GIT_CONFIG hooksPath still chains the repository git is actually in" } +test_empty_project_hookspath_runs_no_repository_hook() { + local repo hooks err + repo="$TMP_ROOT/empty-hookspath" + make_repo "$repo" + write_marker_hook "$repo/.git/hooks/pre-commit" default-pre-commit + git -C "$repo" config core.hooksPath '' + hooks="$TMP_ROOT/hooks-empty" + "$STRIP" install "$hooks" "$repo" || fail "install should succeed with an empty core.hooksPath" + printf 'note\n' >>"$repo/README.md" + git -C "$repo" add README.md + err=$(with_hooks_env "$hooks" git -C "$repo" commit -q --trailer 'Co-authored-by: Cursor ' -m 'fix: empty hooksPath' 2>&1) || + fail "a commit in a repo with an empty core.hooksPath was refused: $err" + assert_equals "" "$err" "an empty core.hooksPath commit printed errors" + [ -f "$repo/default-pre-commit.ran" ] && fail "a repository hook ran although core.hooksPath is empty" + assert_not_contains "$(git -C "$repo" log -1 --format=%B)" "Co-authored-by: Cursor" \ + "Cursor trailer survived an empty-hooksPath commit" + pass "an empty project core.hooksPath runs no repository hook and still strips the trailer" +} + +# git reaches the wrapper's own hooks lookup only where it resolves +# core.hooksPath lazily, at hook lookup, so the pane's GIT_CONFIG override +# supersedes the repository's broken value for git's own commands. Older git +# expands every core.* path as it parses config instead, so a repository whose +# core.hooksPath cannot be resolved refuses EVERY command - `git status` +# included, with or without the override - and the two cases below can neither +# build their fixture nor reach the wrapper. There the property they protect is +# enforced by git itself: no repository hook is silently skipped when no command +# runs at all. Probe the live git with the exact breakage each case uses, rather +# than gating on a version number. +break_hookspath_unresolvable() { # + git -C "$1" config core.hooksPath '~fm-no-such-user-6171/hooks' +} + +break_hookspath_valueless() { # + printf '[core]\n\thooksPath\n' >>"$1/.git/config" +} + +hookspath_break_still_leaves_git_usable() { # + local probe="$TMP_ROOT/hookspath-probe-$1" + rm -rf "$probe" + fm_git_init_commit "$probe" >/dev/null 2>&1 || fail "the core.hooksPath probe repo could not be built" + "$1" "$probe" || fail "the core.hooksPath probe could not break core.hooksPath" + git -C "$probe" rev-parse --is-inside-work-tree >/dev/null 2>&1 +} + +test_unresolvable_project_hookspath_still_refuses() { + local repo hooks head err + hookspath_break_still_leaves_git_usable break_hookspath_unresolvable || { + pass "an unresolvable project core.hooksPath still refuses the commit (skipped: this git refuses every command in such a repository)" + return 0 + } + repo="$TMP_ROOT/unresolvable-hookspath" + make_repo "$repo" + printf 'note\n' >>"$repo/README.md" + git -C "$repo" add README.md + break_hookspath_unresolvable "$repo" + hooks="$TMP_ROOT/hooks-unresolvable" + "$STRIP" install "$hooks" "$repo" || fail "install should succeed with an unresolvable core.hooksPath" + head=$(git -C "$repo" rev-parse HEAD) + err=$(with_hooks_env "$hooks" git -C "$repo" commit -q -m 'fix: unresolvable hooksPath' 2>&1) && + fail "a commit succeeded although the repository's hooks directory cannot be resolved" + assert_contains "$err" "refusing to skip its pre-commit hook" "the refusal did not name the skipped hook" + assert_equals 1 "$(printf '%s\n' "$err" | grep -c 'failed to expand user dir')" "git's lookup error was not shown exactly once" + assert_equals "$head" "$(git -C "$repo" rev-parse HEAD)" "a refused commit still moved HEAD" + pass "an unresolvable project core.hooksPath still refuses the commit" +} + +test_valueless_project_hookspath_still_refuses() { + local repo hooks head err + hookspath_break_still_leaves_git_usable break_hookspath_valueless || { + pass "a valueless project core.hooksPath still refuses the commit (skipped: this git refuses every command in such a repository)" + return 0 + } + repo="$TMP_ROOT/valueless-hookspath" + make_repo "$repo" + hooks="$TMP_ROOT/hooks-valueless" + "$STRIP" install "$hooks" "$repo" || fail "install should succeed before the valueless key is written" + head=$(git -C "$repo" rev-parse HEAD) + printf 'note\n' >>"$repo/README.md" + git -C "$repo" add README.md + break_hookspath_valueless "$repo" + err=$(with_hooks_env "$hooks" git -C "$repo" commit -q -m 'fix: valueless hooksPath' 2>&1) && + fail "a commit succeeded although core.hooksPath has no value" + assert_contains "$err" "refusing to skip its pre-commit hook" "the refusal did not name the skipped hook" + assert_equals 1 "$(printf '%s\n' "$err" | grep -c "missing value for 'core.hookspath'")" "git's lookup error was not shown exactly once" + assert_equals "$head" "$(git -C "$repo" -c core.hooksPath=x rev-parse HEAD)" "a refused commit still moved HEAD" + pass "a valueless project core.hooksPath still refuses the commit" +} + write_refusing_pre_push() { # cat >"$1" < + local state + state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" + printf "export FM_TASK_INBOX='%s'; " "$state/$2.inbox" +} + ai_trailer_hooks_prefix() { # local state state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" @@ -300,7 +306,7 @@ test_kimi_launch_then_send_is_verified() { assert_contains "$out" "spawned $id harness=kimi" "kimi spawn did not report success" launch=$(cat "$CASE_DIR/launch.log") - [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$FAKEBIN_DIR/kimi' --model 'kimi-code/k3' --auto" ] \ + [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$HOME_DIR" "$id")$(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$FAKEBIN_DIR/kimi' --model 'kimi-code/k3' --auto" ] \ || fail "kimi launch did not use the absolute binary, model, and --auto only: $launch" assert_not_contains "$launch" "--effort" "kimi launch emitted a nonexistent effort flag" assert_not_contains "$launch" "turn-ended" "kimi launch embedded a turn-end path" @@ -676,7 +682,7 @@ test_kimi_falls_back_to_expanded_home_binary() { rc=$? expect_code 0 "$rc" "Kimi HOME fallback spawn should succeed" launch=$(cat "$CASE_DIR/launch.log") - [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$fallback' --auto" ] \ + [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$HOME_DIR" "$id")$(ai_trailer_hooks_prefix "$HOME_DIR" "$id")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI '$fallback' --auto" ] \ || fail "Kimi fallback did not expand HOME into an absolute executable: $launch" pass "fm-spawn: Kimi fallback expands the active HOME" } diff --git a/tests/fm-send-inbox-doorbell-live-e2e.test.sh b/tests/fm-send-inbox-doorbell-live-e2e.test.sh index 36de34f56b4..0b1240ddddf 100644 --- a/tests/fm-send-inbox-doorbell-live-e2e.test.sh +++ b/tests/fm-send-inbox-doorbell-live-e2e.test.sh @@ -4,14 +4,17 @@ # # The steering inbox's one behavioral assumption is that a real worker agent # follows the constant self-describing doorbell line: list the inbox, read and -# act on its records in numeric order, then mv each into handled/. A stub can -# only confirm the assumption already -# written into the stub, so per .agents/skills/firstmate-coding-guidelines -# this is proven against every INSTALLED verified harness: each is launched -# idle in an isolated tmux server, steered through the REAL fm-send (durable -# record + doorbell), and must both ACT on the instruction (create a named -# file) and ACKNOWLEDGE it (the mv into handled/), failing loudly with the -# harness name and version. +# act on its records in numeric order, then mv each into handled/. The +# doorbell names the inbox as "$FM_TASK_INBOX", so each worker is launched the +# way bin/fm-spawn.sh launches it, with FM_TASK_INBOX exported to its home's +# state/.inbox, and receives no brief at all: it must resolve the inbox +# from the doorbell plus its own environment. A stub can only confirm the +# assumption already written into the stub, so per +# .agents/skills/firstmate-coding-guidelines this is proven against every +# INSTALLED verified harness: each is launched idle in an isolated tmux server, +# steered through the REAL fm-send (durable record + doorbell), and must both +# ACT on the instruction (create a named file) and ACKNOWLEDGE it (the mv into +# handled/), failing loudly with the harness name and version. # # Run explicitly with FM_SEND_INBOX_LIVE_E2E=1. This test spends a small # number of real model tokens per installed harness (one short turn each) - @@ -131,7 +134,7 @@ check_harness_doorbell() { # task="live-$name" acted="$LAB/acted-$name" tmux -L "$SOCKET" new-window -d -t "$SESSION:" -n "$win" -c "$ROOT" \ - -- bash -lc "$cmd" \ + -- bash -lc "export FM_TASK_INBOX=$(printf '%q' "$home/state/$task.inbox"); $cmd" \ || { FAILED=1; printf 'not ok - %s (%s): could not launch in the isolated tmux server\n' "$name" "$version" >&2; return 0; } wait_ready "$win"; ready_rc=$? if [ "$ready_rc" -eq 1 ]; then diff --git a/tests/fm-send-inbox.test.sh b/tests/fm-send-inbox.test.sh index b669a6d9860..c0c61f62ae5 100644 --- a/tests/fm-send-inbox.test.sh +++ b/tests/fm-send-inbox.test.sh @@ -7,6 +7,7 @@ # drive the real fm-send executable over a stubbed tmux and pin: # 1. The payload is durably recorded and never typed; only the doorbell # crosses the terminal, and the send exits 0 at enqueue. +# The doorbell names the inbox once and never grows with the home's depth. # 2. Multi-line steers are legal and round-trip byte-exact. # 3. A re-send enqueues a NEW sequence and still never retypes a payload, # so the terminal can never truncate, garble, or duplicate a steer. @@ -129,7 +130,7 @@ test_text_steer_rides_inbox() { body=$(record_body _ "$rec") [ "$body" = "please rebase onto main" ] || fail "the recorded body differs: $body" typed=$(cat "$dir/send.log") - assert_contains "$typed" "Firstmate instruction waiting: list '$dir/home/state/t1.inbox'/*.msg" \ + assert_contains "$typed" "Firstmate instruction waiting: list \"\$FM_TASK_INBOX\"/*.msg in your 't1.inbox' steering inbox" \ "the doorbell should direct the worker to drain the inbox" case "$typed" in *"please rebase onto main"*) fail "the payload must never be typed:"$'\n'"$typed" ;; @@ -137,6 +138,47 @@ test_text_steer_rides_inbox() { pass "fm-send inbox: the payload is recorded durably and only the doorbell is typed" } +# A home nested deep must not lengthen the doorbell: a long line wraps past +# what a composer read can prove, so a Herdr submit reports it never reached +# the pane and every re-ring fails the same way. +test_deep_home_doorbell_stays_short() { + local shallow deep home err rest typed shallow_typed found + shallow=$(setup_case shallow-home) + run_send "$shallow" "$shallow/send.err" -- t1 "please continue" || fail "the shallow-home send failed" + shallow_typed=$(cat "$shallow/send.log") + deep="$TMP_ROOT/deep-home" + home="$deep/one/two/three/four/five/six/seven/eight-secondmate-homes-nest-under-long-worktree-paths" + mkdir -p "$home/state" + make_stubs "$deep" >/dev/null + fm_write_meta "$home/state/t1.meta" "window=sess:fm-t1" "kind=ship" "harness=claude" + err="$deep/send.err" + env PATH="$deep/fakebin:$PATH" \ + FM_ROOT_OVERRIDE="$home" FM_HOME="$home" FM_SEND_LOG="$deep/send.log" \ + FM_SEND_SETTLE=0 "$SEND" t1 "please continue" >/dev/null 2>"$err" || + fail "the deep-home send failed: $(cat "$err")" + [ -f "$home/state/t1.inbox/001.msg" ] || fail "the deep-home steer was not durably recorded" + typed=$(cat "$deep/send.log") + [ "$typed" = "$shallow_typed" ] || + fail "the doorbell should not depend on the home's depth:"$'\n'"shallow: $shallow_typed"$'\n'"deep: $typed" + [ "${#typed}" -le 200 ] || fail "the doorbell should stay under 200 characters, got ${#typed}: $typed" + case "$typed" in + *"$deep"* | *"$TMP_ROOT"*) fail "the doorbell should not carry the home's absolute path: $typed" ;; + esac + rest=${typed#*t1.inbox} + [ "$rest" != "$typed" ] || fail "the doorbell should name the inbox: $typed" + case "$rest" in + *t1.inbox*) fail "the doorbell should name the inbox once: $typed" ;; + esac + found=$(cd / && FM_TASK_INBOX="$home/state/t1.inbox" bash -c 'ls "$FM_TASK_INBOX"/*.msg') || + fail "a shell with FM_TASK_INBOX exported could not list the deep inbox" + [ "$found" = "$home/state/t1.inbox/001.msg" ] || + fail "the doorbell's list instruction did not resolve the deep inbox from an unrelated cwd: $found" + (cd / && FM_TASK_INBOX="$home/state/t1.inbox" bash -c 'mv "$FM_TASK_INBOX"/001.msg "$FM_TASK_INBOX"/handled/') || + fail "the doorbell's mv instruction did not acknowledge through FM_TASK_INBOX" + [ -f "$home/state/t1.inbox/handled/001.msg" ] || fail "the acknowledged record did not land in handled/" + pass "fm-send inbox: a deep home rings the same short doorbell naming the inbox once" +} + test_multiline_steer_is_legal() { local dir err rc body dir=$(setup_case multiline) @@ -412,6 +454,7 @@ test_empty_message_refused() { } test_text_steer_rides_inbox +test_deep_home_doorbell_stays_short test_multiline_steer_is_legal test_resend_enqueues_new_sequence test_pending_composer_skips_ring_advisorily diff --git a/tests/fm-session-start.test.sh b/tests/fm-session-start.test.sh index 2e76714a07e..327663cfbb1 100755 --- a/tests/fm-session-start.test.sh +++ b/tests/fm-session-start.test.sh @@ -1491,7 +1491,11 @@ EOF printf 'window=sess:p-slow\nkind=ship\nbackend=herdr\n' > "$home/state/task-a-slow.meta" printf 'window=sess:p-live\nkind=ship\nbackend=herdr\n' > "$home/state/task-z-live.meta" - out=$(FM_SESSION_START_ENDPOINT_TIMEOUT=2 run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + # The same fake hangs the side-band home summary before the endpoint section. + # Bound that unrelated refresh at 5s instead of paying its production 60s; + # the endpoint's own 2s bound and descendant-cleanup assertions stay real. + out=$(FM_HOME_SUMMARY_TIMEOUT=5 FM_SESSION_START_ENDPOINT_TIMEOUT=2 \ + run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? expect_code 0 "$status" "a hung endpoint read must not fail the digest" assert_contains "$out" \ @@ -1527,7 +1531,10 @@ EOF printf 'window=sess:p-slow\nkind=ship\nbackend=herdr\n' > "$home/state/task-a-slow.meta" printf 'window=sess:p-live\nkind=ship\nbackend=herdr\n' > "$home/state/task-z-live.meta" - out=$(FM_SESSION_START_ENDPOINT_TIMEOUT=00 run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + # Only the unrelated summary gets a shorter fixture budget. The invalid + # endpoint value must still fall back to the real 10s production bound. + out=$(FM_HOME_SUMMARY_TIMEOUT=5 FM_SESSION_START_ENDPOINT_TIMEOUT=00 \ + run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? expect_code 0 "$status" "a padded-zero per-read bound must not fail the digest" assert_contains "$out" \ diff --git a/tests/fm-spawn-compact-adviser-disable.test.sh b/tests/fm-spawn-compact-adviser-disable.test.sh index d2713604caf..f9b7f7482d1 100755 --- a/tests/fm-spawn-compact-adviser-disable.test.sh +++ b/tests/fm-spawn-compact-adviser-disable.test.sh @@ -59,10 +59,10 @@ run_case_spawn() { # Replace the harness binary with a probe that reports the single environment # fact under test, so executing the emitted launch answers "what would the agent # have seen" rather than "what does the command text look like". -install_env_probe() { # - cat > "$1/$2" <<'SH' +install_env_probe() { # [variable] + cat > "$1/$2" < "$HOME_DIR/config/launch-env-allowlist" + if [ "$kind" = ship ]; then + out=$(run_case_spawn "$id" "$PROJ_DIR" --mode no-mistakes --yolo off) + else + sm="$CASE_DIR/secondmate-home" + mkdir -p "$sm/bin" "$sm/data" + printf '# Firstmate\n' > "$sm/AGENTS.md" + printf '%s\n' "$id" > "$sm/.fm-secondmate-home" + printf 'charter for %s\n' "$id" > "$sm/data/charter.md" + printf '%s\n' 'projects/' 'state/' 'data/' 'config/' '.no-mistakes/' > "$sm/.gitignore" + git -C "$sm" init -q -b main + out=$(run_case_spawn "$id" "$sm" --secondmate) + fi + status=$? + expect_code 0 "$status" "$kind spawn should succeed: $out" + install_env_probe "$FAKEBIN_DIR" codex FM_TASK_INBOX + seen=$(emitted_launch_env "$FAKEBIN_DIR" "$LAUNCH_LOG" "$PANE_LOG") \ + || fail "$kind: the emitted launch failed to run" + want="$(cd "$HOME_DIR/state" && pwd -P)/$id.inbox" + assert_equals "$want" "$seen" \ + "a $kind agent must start with FM_TASK_INBOX set to its absolute steering inbox" + done + pass "ship and secondmate launches export their absolute steering inbox as FM_TASK_INBOX" +} + # --- relaunch --------------------------------------------------------------- # # bin/fm-control.sh relaunch stops the agent and rebuilds the launch through @@ -351,5 +386,6 @@ test_ship_allowlist_absent test_ship_allowlist_enabled test_launch_command_carries_the_switch_without_the_pane_export test_secondmate_launch +test_launch_exports_task_inbox test_relaunch_rebuilds_the_switch test_raw_compound_launch_command_carries_the_switch diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 7d20e674d06..cc9e302e092 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -87,6 +87,12 @@ make_seeded_secondmate_home() { git -C "$home" init -q -b main } +task_inbox_export() { # + local state + state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" + printf "export FM_TASK_INBOX='%s'; " "$state/$2.inbox" +} + ai_trailer_hooks_prefix() { # local state state=$(CDPATH='' cd -- "$1/state" && pwd -P) || fail "cannot resolve state dir $1/state" @@ -476,7 +482,7 @@ test_active_dispatch_profile_allows_raw_launch_command() { # The unverified-adapter escape hatch is still an agent this fleet launched, # so it carries the compact-adviser floor and the AI-trailer strip; nothing # else may rewrite the captain's own command. - [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$HOME_DIR" "$id")custom-agent --flag" ] || fail "raw launch command changed"$'\n'"actual: $launch" + [ "$launch" = "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$HOME_DIR" "$id")$(ai_trailer_hooks_prefix "$HOME_DIR" "$id")custom-agent --flag" ] || fail "raw launch command changed"$'\n'"actual: $launch" pass "active crew-dispatch profile allows the raw launch-command escape hatch" } @@ -2009,7 +2015,7 @@ claude_expected_launch() { # [ "$(printf '%s' "$doorbell" | "$ROOT/bin/fm-operational-input.sh" doorbell-kind)" = launch-brief ] \ || doorbell="not a launch-brief doorbell" quoted="'$(printf '%s' "$doorbell" | sed "s/'/'\\\\''/g")'" - printf '%s' "export COMPACT_ADVISER_DISABLE=1; $(ai_trailer_hooks_prefix "$2" "$3")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude $4 $(claude_worker_add_dirs "$2" "$3")--settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' $CLAUDE_CONTROL_CHANNEL_FLAG $quoted" + printf '%s' "export COMPACT_ADVISER_DISABLE=1; $(task_inbox_export "$2" "$3")$(ai_trailer_hooks_prefix "$2" "$3")env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude $4 $(claude_worker_add_dirs "$2" "$3")--settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' $CLAUDE_CONTROL_CHANNEL_FLAG $quoted" } test_claude_permission_mode_bypass_matches_absent_launch() { diff --git a/tests/fm-startup-memory-budget.test.sh b/tests/fm-startup-memory-budget.test.sh index 6e53816c024..fa4a6daca22 100755 --- a/tests/fm-startup-memory-budget.test.sh +++ b/tests/fm-startup-memory-budget.test.sh @@ -280,7 +280,7 @@ test_primary_budget_converges_with_exact_reread_and_safe_failures() { "budget propagation did not enqueue the pointer to its exact reread generation" assert_contains "$(<"$log")" "Firstmate instruction waiting: list " \ "budget propagation did not ring the durable inbox doorbell" - assert_contains "$(<"$log")" "/state/sm.inbox'/*.msg" \ + assert_contains "$(<"$log")" "'sm.inbox' steering inbox" \ "budget propagation doorbell did not identify the durable inbox" outside="$world/unsafe-budget" diff --git a/tests/fm-task-inbox.test.sh b/tests/fm-task-inbox.test.sh index eda6f37170c..7a188c76422 100644 --- a/tests/fm-task-inbox.test.sh +++ b/tests/fm-task-inbox.test.sh @@ -162,9 +162,10 @@ test_write_is_durable_and_exact() { doorbell2=$(inbox_lib "$state" fm_task_inbox_doorbell_line "$rec2") [ "$doorbell" = "$doorbell2" ] \ || fail "every record in one inbox should ring the same drain-all doorbell" - assert_contains "$doorbell" "'$state/t1.inbox'/*.msg" "doorbell should quote and name all unhandled records" + assert_contains "$doorbell" "list \"\$FM_TASK_INBOX\"/*.msg" "doorbell should list all unhandled records through FM_TASK_INBOX" + assert_contains "$doorbell" "'t1.inbox' steering inbox" "doorbell should quote and name the inbox" assert_contains "$doorbell" "numeric order" "doorbell should require ordered processing" - assert_contains "$doorbell" "'$state/t1.inbox'/handled/" "doorbell should quote and name the handled dir" + assert_contains "$doorbell" "handled/" "doorbell should name the handled dir" assert_contains "$doorbell" "Firstmate instruction waiting" "doorbell should be self-describing" case "$doorbell" in *$'\n'*) fail "the doorbell must be a single line" ;; @@ -181,69 +182,70 @@ test_write_is_durable_and_exact() { # command line. Execute the real line in real shells and assert it is inert: # exit 0, no output, and nothing in the inbox touched. test_doorbell_is_a_shell_noop() { - local state rec doorbell sh out before after marker - state="$TMP_ROOT/noop/x; touch marker; #'s space/state" + local state task rec doorbell sh out before after marker + state="$TMP_ROOT/noop/state" + task="x; touch marker; #'s space" marker="$state/marker" mkdir -p "$state" - rec=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "please continue") + rec=$(inbox_lib "$state" fm_task_inbox_write "$state" "$task" "please continue") doorbell=$(inbox_lib "$state" fm_task_inbox_doorbell_line "$rec") case "$doorbell" in ': '*) ;; *) fail "the doorbell must start with the shell no-op prefix, got: $doorbell" ;; esac - assert_contains "$doorbell" "'\\''s space/state/t1.inbox'" \ - "the doorbell should escape an embedded single quote in its quoted path" - before=$(ls -R "$state/t1.inbox") + assert_contains "$doorbell" "'\\''s space.inbox'" \ + "the doorbell should escape an embedded single quote in its quoted inbox name" + before=$(ls -R "$state/$task.inbox") for sh in sh bash zsh; do command -v "$sh" >/dev/null 2>&1 || continue - out=$(cd "$state" && "$sh" -c "$doorbell" 2>&1) \ + out=$(cd "$state" && FM_TASK_INBOX="$state/$task.inbox" "$sh" -c "$doorbell" 2>&1) \ || fail "$sh executed the hostile-path doorbell with a non-zero status: $out" [ -z "$out" ] || fail "$sh produced output while executing the hostile-path doorbell: $out" - [ ! -e "$marker" ] || fail "$sh executed shell syntax embedded in the inbox path" + [ ! -e "$marker" ] || fail "$sh executed shell syntax embedded in the inbox name" done # An interactive-style zsh with the line fed on stdin, the closest portable # stand-in for a dead pane's login shell reading typed keystrokes. if command -v zsh >/dev/null 2>&1; then - out=$(cd "$state" && printf '%s\n' "$doorbell" | zsh -s 2>&1) \ + out=$(cd "$state" && printf '%s\n' "$doorbell" | FM_TASK_INBOX="$state/$task.inbox" zsh -s 2>&1) \ || fail "zsh reading the hostile-path doorbell from stdin failed: $out" [ -z "$out" ] || fail "zsh printed while reading the hostile-path doorbell: $out" [ ! -e "$marker" ] || fail "zsh executed shell syntax from the stdin doorbell" fi - after=$(ls -R "$state/t1.inbox") + after=$(ls -R "$state/$task.inbox") [ "$before" = "$after" ] || fail "executing the doorbell changed the inbox:"$'\n'"$after" [ -f "$rec" ] || fail "executing the doorbell removed the unhandled record" - pass "inbox: a hostile-path doorbell executes as a no-op in bare shells" + pass "inbox: a hostile-name doorbell executes as a no-op in bare shells" } test_doorbell_rejects_terminal_controls() { - local dir state rec doorbell control label log marker rc + local dir state task rec doorbell control label log marker rc dir="$TMP_ROOT/control-path" + state="$dir/state" marker="$dir/marker" - mkdir -p "$dir" + mkdir -p "$state" make_watch_stubs "$dir" >/dev/null for label in etx esc; do case "$label" in etx) control=$'\003' ;; esc) control=$'\033' ;; esac - state="$dir/${control}touch marker; # $label/state" - mkdir -p "$state" - rec=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "please continue") + task="${control}touch marker; # $label" + rec=$(inbox_lib "$state" fm_task_inbox_write "$state" "$task" "please continue") doorbell= rc=0 doorbell=$(inbox_lib "$state" fm_task_inbox_doorbell_line "$rec") || rc=$? - [ "$rc" -ne 0 ] || fail "a $label path should make doorbell construction fail" - [ -z "$doorbell" ] || fail "a rejected $label path emitted doorbell bytes" + [ "$rc" -ne 0 ] || fail "a $label inbox name should make doorbell construction fail" + [ -z "$doorbell" ] || fail "a rejected $label inbox name emitted doorbell bytes" log="$dir/$label.send.log"; : > "$log" rc=0 PATH="$dir/fakebin:$PATH" FM_SEND_LOG="$log" \ inbox_lib "$state" fm_task_inbox_ring tmux sess:fm-t1 "$rec" fm-t1 || rc=$? - [ "$rc" = 2 ] || fail "a rejected $label path should return send-failed status 2, got $rc" - [ ! -s "$log" ] || fail "a $label path reached send-keys:"$'\n'"$(cat "$log")" - [ ! -e "$marker" ] || fail "a $label path executed its crafted command" - [ -f "$rec" ] || fail "rejecting a $label path removed the durable record" + [ "$rc" = 2 ] || fail "a rejected $label inbox name should return send-failed status 2, got $rc" + [ ! -s "$log" ] || fail "a $label inbox name reached send-keys:"$'\n'"$(cat "$log")" + [ ! -e "$marker" ] || fail "a $label inbox name executed its crafted command" + [ -f "$rec" ] || fail "rejecting a $label inbox name removed the durable record" done - pass "inbox: terminal-control paths are rejected without typing" + pass "inbox: terminal-control inbox names are rejected without typing" } # fm_task_inbox_ring against a backend whose agent classifies dead or missing: @@ -616,7 +618,7 @@ test_watcher_rerings_idle_pane_quietly() { sleep 0.1 i=$((i + 1)) done - grep -qF "Firstmate instruction waiting: list '$state/t1.inbox'/*.msg" "$log" \ + grep -qF "Firstmate instruction waiting: list \"\$FM_TASK_INBOX\"/*.msg in your 't1.inbox' steering inbox" "$log" \ || { kill "$pid" 2>/dev/null; fail "the watcher never re-rang the doorbell:"$'\n'"$(cat "$log")"; } kill -0 "$pid" 2>/dev/null \ || fail "a healthy re-ring must not wake firstmate (watcher exited):"$'\n'"$(cat "$out")" diff --git a/tests/fm-teardown-endpoint-safety.test.sh b/tests/fm-teardown-endpoint-safety.test.sh index b8f8ac201d7..049ee3b09cf 100755 --- a/tests/fm-teardown-endpoint-safety.test.sh +++ b/tests/fm-teardown-endpoint-safety.test.sh @@ -1024,6 +1024,43 @@ test_reassigned_pool_slot_finishes_own_cleanup_without_touching_the_slot() { pass "fm-teardown: a pool slot claimed by another task is left alone while the task's own cleanup finishes" } +# The reuse collision where BOTH records survive: the stale task's record still +# names the slot the pool handed on, and the claimant's own record names it too. +# The claim proves the stale record's teardown is records-only, so the record +# scan must not refuse it; once it is gone, the claimant tears down normally. +test_stale_record_on_claimed_slot_retires_then_claimant_tears_down() { + local dir id=stale-task other=live-task rc + + dir=$(make_case slot-reassigned-both-records) + mark_case_as_treehouse_pool "$dir" + fm_write_meta "$dir/home/state/$id.meta" \ + "window=firstmate:fm-$id" "endpoint_task_id=$id" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + fm_write_meta "$dir/home/state/$other.meta" \ + "window=firstmate:fm-$other" "endpoint_task_id=$other" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + claim_pool_slot "$dir" "$other" + + set +e + run_case "$dir" "$id" > "$dir/stdout" 2> "$dir/stderr" + rc=$? + set -e + [ "$rc" -eq 0 ] || fail "records-only teardown of a stale record on a claimed slot failed: $(cat "$dir/stderr")" + assert_reassigned_slot_left_alone "$dir" "$id" "$other" "stale record beside the claimant's record" + assert_present "$dir/worktree/sentinel" "records-only teardown reset the claimant's slot" + assert_present "$dir/home/state/$other.meta" "records-only teardown removed the claimant's record" + + : > "$dir/runtime.log" + run_case "$dir" "$other" > "$dir/stdout" 2> "$dir/stderr" \ + || fail "claimant teardown failed after the stale record retired: $(cat "$dir/stderr")" + assert_absent "$dir/home/state/$other.meta" "claimant teardown left its record" + assert_absent "$dir/pool/1/.fm-slot-owner" "claimant teardown left its spent slot claim behind" + grep -Fq "treehouse " "$dir/runtime.log" \ + || fail "claimant teardown did not return its pool slot: $(cat "$dir/runtime.log")" + + pass "fm-teardown: a stale record on a claimed slot retires, then the claimant tears down" +} + # The two states that must never become a false refusal: the task's own claim, # and no claim at all (a slot taken before claims existed, or already returned). test_own_and_absent_slot_claims_still_tear_down() { @@ -1444,6 +1481,7 @@ test_reused_pool_slot_refuses_before_touching_the_other_task test_cross_home_pool_slot_collision_refuses test_sole_slot_record_still_tears_down test_reassigned_pool_slot_finishes_own_cleanup_without_touching_the_slot +test_stale_record_on_claimed_slot_retires_then_claimant_tears_down test_own_and_absent_slot_claims_still_tear_down test_recorded_endpoint_that_changed_directory_still_tears_down test_project_lock_anchors_at_the_local_root_across_home_layouts diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index 211a6fde79c..924342c9f06 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -1028,11 +1028,11 @@ test_list_scheduled_non_lane_selections_use_serial_weights() { printf '\n' >>"$repo/$script" done printf '%s\n' \ + tests/fm-kimi-harness.test.sh \ tests/fm-muse-harness.test.sh \ tests/fm-brief.test.sh \ tests/fm-captain-hold-lifecycle.test.sh \ tests/fm-lint.test.sh \ - tests/fm-kimi-harness.test.sh \ tests/fm-operational-input.test.sh >"$tmp/expected" for selection in family all changed scripts; do case "$selection" in @@ -1157,7 +1157,7 @@ test_portable_serial_shards_partition_the_serial_lane() { } test_portable_serial_hint_coverage_is_reported_and_bounded() { - local out serial unhinted + local out serial unhinted max budget # Shards are packed from measured duration hints, so an unmeasured script is # placed on a guess. Enough of them and the partition still looks balanced by # script count while one shard carries far more real work than another and @@ -1178,7 +1178,56 @@ test_portable_serial_hint_coverage_is_reported_and_bounded() { # this trips (docs/fm-test-portable-shards.md). [ "$((unhinted * 100))" -le "$((serial * 15))" ] \ || fail "$unhinted of $serial portable serial scripts lack a measured hint; refresh them" - pass "coverage guard reports and bounds the unmeasured portable serial share" + # A complete partition can still overflow a CI job. Assert the runner's + # modeled packing target through its executable interface, not source hints. + max=$(printf '%s\n' "$out" | sed -n 's/.*serial_max_ms=\([0-9][0-9]*\).*/\1/p') + budget=$(printf '%s\n' "$out" | sed -n 's/.*serial_budget_ms=\([0-9][0-9]*\).*/\1/p') + [ -n "$max" ] && [ -n "$budget" ] \ + || fail "coverage summary must carry serial packing and budget: $out" + [ "$budget" -eq 1200000 ] || fail "packing must leave forty minutes of the normal CI tier" + [ "$max" -gt 0 ] && [ "$max" -le "$budget" ] \ + || fail "largest serial shard packs ${max}ms above the ${budget}ms target" + pass "coverage guard bounds the unmeasured share and serial packing within twenty minutes" +} + +test_portable_serial_packing_budget_boundary() { + local tmp repo script weight out rc + tmp=$(fm_test_tmproot fm-test-run-packing-boundary) + repo="$tmp/repo" + mkdir -p "$repo/bin" "$repo/tests" + # Preserve the real inventory and packing policy without executing suites. + # Only the fixture's measured timing input changes at the boundary. + while IFS= read -r script; do + printf '#!/usr/bin/env bash\nexit 0\n' >"$repo/$script" + done < <("$RUNNER" --list --all) + + for weight in 1200000 1200001; do + cp "$RUNNER" "$repo/bin/fm-test-run.sh" + python3 - "$repo/bin/fm-test-run.sh" "$weight" <<'PY' \ + || fail "could not seed the fixture's measured timing input" +from pathlib import Path +import re, sys +runner = Path(sys.argv[1]) +runner.write_text(re.sub( + r"(?m)^tests/fm-watch-triage\.test\.sh [0-9]+$", + f"tests/fm-watch-triage.test.sh {sys.argv[2]}", + runner.read_text(), +)) +PY + out=$(bash "$repo/bin/fm-test-run.sh" --check-coverage 2>&1) && rc=0 || rc=$? + if [ "$weight" -eq 1200000 ]; then + expect_code 0 "$rc" "packing exactly at the budget must be accepted" + assert_contains "$out" "FM_TEST_COVERAGE ok" "boundary coverage did not pass" + assert_contains "$out" "serial_max_ms=1200000" "fixture did not pack exactly at the budget" + assert_contains "$out" "serial_budget_ms=1200000" "fixture changed the packing budget" + else + expect_code 1 "$rc" "packing one millisecond above the budget must be refused" + assert_contains "$out" "largest portable serial shard packs 1200001ms above the 1200000ms target" \ + "over-budget refusal did not explain the modeled excess" + assert_not_contains "$out" "FM_TEST_COVERAGE ok" "over-budget packing reported success" + fi + done + pass "serial packing accepts the exact budget and refuses one millisecond above it" } test_portable_serial_shard_lane_refusals() { @@ -1805,6 +1854,7 @@ test_portable_shard_union_and_coverage_guard test_portable_parallel_lanes_stay_duration_balanced test_portable_serial_shards_partition_the_serial_lane test_portable_serial_hint_coverage_is_reported_and_bounded +test_portable_serial_packing_budget_boundary test_portable_serial_shard_lane_refusals test_jobs_requires_proven_isolated test_jobs_admits_a_concurrent_safe_family