From 3e46fbde9d0dccd7ba6287c1968ab0e6c96678e6 Mon Sep 17 00:00:00 2001 From: dnth Date: Mon, 5 Oct 2026 21:59:18 +0800 Subject: [PATCH 1/2] fix(bin): pin remote secondmate relaunch to the herdr backend cmd_relaunch omitted the --backend herdr that cmd_launch passes, so a remote relaunch resolved the remote home's configured backend (tmux), failed OMP session-lock binding, and stranded the route. fm-control relaunch now accepts --backend, forwards it to the launch owner, and refuses before touching anything when the recorded endpoint lives on a different backend. --- bin/fm-control.sh | 25 ++++- bin/fm-remote-secondmate-control.sh | 6 +- docs/remote-secondmates.md | 2 +- ...fm-remote-secondmate-lifecycle-e2e.test.sh | 61 ++++++++++++ tests/fm-spawn-relaunch-dead-endpoint.test.sh | 94 +++++++++++++++++++ 5 files changed, 183 insertions(+), 5 deletions(-) diff --git a/bin/fm-control.sh b/bin/fm-control.sh index 13d8e962a4e..bb32e762557 100755 --- a/bin/fm-control.sh +++ b/bin/fm-control.sh @@ -5,7 +5,8 @@ # Usage: fm-control.sh interrupt # fm-control.sh exit # fm-control.sh relaunch [--harness ] [--model ] -# [--effort ] [--lock-preheld] +# [--effort ] [--backend ] +# [--lock-preheld] # (--note | --note-file ) # # Why this exists, and how it differs from fm-send.sh. bin/fm-send.sh is the @@ -46,6 +47,10 @@ # inherits the local copy but none of the conversation; a # secondmate reconciles its own home's records at startup, so its # standing charter is never rewritten. +# --backend pins the replacement's runtime backend instead of +# letting the launch owner resolve it from this home's config; a +# task whose recorded endpoint sits on a different backend is +# refused before anything is touched. # --lock-preheld is the supervised-recovery handshake: the caller # (bin/fm-stall-recovery.sh) already holds this task's lifecycle # lock, so fm-control verifies the lock's recorded owner is its @@ -211,9 +216,11 @@ fi NEW_HARNESS= NEW_MODEL= NEW_EFFORT= +PIN_BACKEND= HARNESS_SET=0 MODEL_SET=0 EFFORT_SET=0 +BACKEND_SET=0 NOTE= NOTE_SET=0 LOCK_PREHELD=0 @@ -228,6 +235,7 @@ for control_arg in "$@"; do harness) NEW_HARNESS=$control_arg; HARNESS_SET=1 ;; model) NEW_MODEL=$control_arg; MODEL_SET=1 ;; effort) NEW_EFFORT=$control_arg; EFFORT_SET=1 ;; + backend) PIN_BACKEND=$control_arg; BACKEND_SET=1 ;; note) NOTE=$control_arg; NOTE_SET=1 ;; stall_record) STALL_RECORD=$control_arg ;; note_file) @@ -246,6 +254,8 @@ for control_arg in "$@"; do --model=*) NEW_MODEL=${control_arg#--model=}; MODEL_SET=1 ;; --effort) control_want_value=effort ;; --effort=*) NEW_EFFORT=${control_arg#--effort=}; EFFORT_SET=1 ;; + --backend) control_want_value=backend ;; + --backend=*) PIN_BACKEND=${control_arg#--backend=}; BACKEND_SET=1 ;; --note) control_want_value=note ;; --note=*) NOTE=${control_arg#--note=}; NOTE_SET=1 ;; --note-file) control_want_value=note_file ;; @@ -266,8 +276,8 @@ if [ -n "$control_want_value" ]; then fi if [ "$VERB" != relaunch ]; then - [ "$HARNESS_SET" = 0 ] && [ "$MODEL_SET" = 0 ] && [ "$EFFORT_SET" = 0 ] && [ "$NOTE_SET" = 0 ] && [ "$LOCK_PREHELD" = 0 ] && [ -z "$STALL_RECORD" ] \ - || die "--harness, --model, --effort, --note, --lock-preheld, and --stall-record apply to 'relaunch' only" + [ "$HARNESS_SET" = 0 ] && [ "$MODEL_SET" = 0 ] && [ "$EFFORT_SET" = 0 ] && [ "$BACKEND_SET" = 0 ] && [ "$NOTE_SET" = 0 ] && [ "$LOCK_PREHELD" = 0 ] && [ -z "$STALL_RECORD" ] \ + || die "--harness, --model, --effort, --backend, --note, --lock-preheld, and --stall-record apply to 'relaunch' only" fi # The stall-record re-check is only meaningful inside the supervised-recovery # handshake: without --lock-preheld there is no proof the caller serialized @@ -277,6 +287,7 @@ fi [ "$HARNESS_SET" = 0 ] || [ -n "$NEW_HARNESS" ] || die "--harness requires a non-empty value" [ "$MODEL_SET" = 0 ] || [ -n "$NEW_MODEL" ] || die "--model requires a non-empty value" [ "$EFFORT_SET" = 0 ] || [ -n "$NEW_EFFORT" ] || die "--effort requires a non-empty value" +[ "$BACKEND_SET" = 0 ] || [ -n "$PIN_BACKEND" ] || die "--backend requires a non-empty value" case "$NEW_EFFORT" in ''|default|low|medium|high|xhigh|max) ;; *) die "--effort must be one of default, low, medium, high, xhigh, max" ;; @@ -357,6 +368,13 @@ fm_control_harness_supported "$HARNESS" \ fm_backend_validate "$BACKEND" || exit 1 +# A pinned replacement backend must be the backend the recorded endpoint +# already lives on: refusing here, before any journal write or agent touch, +# keeps a mismatched pin from silently migrating the task onto another backend. +if [ "$VERB" = relaunch ] && [ "$BACKEND_SET" = 1 ] && [ "$PIN_BACKEND" != "$BACKEND" ]; then + die "task $ID's endpoint is recorded on backend '$BACKEND', not the pinned '$PIN_BACKEND'; refusing to relaunch it onto another backend" +fi + # --- shared helpers --------------------------------------------------------- # The durable record is part of the question for an adapter whose liveness is @@ -953,6 +971,7 @@ do_relaunch() { else spawn_args=("$ID" --relaunch --harness "$TARGET_HARNESS") fi + [ "$BACKEND_SET" = 0 ] || spawn_args+=(--backend "$PIN_BACKEND") [ "$TARGET_MODEL" = default ] || spawn_args+=(--model "$TARGET_MODEL") [ "$TARGET_EFFORT" = default ] || spawn_args+=(--effort "$TARGET_EFFORT") # FM_RESTART_SPAWN_CMD replaces the launch owner (default: bin/fm-spawn.sh) so diff --git a/bin/fm-remote-secondmate-control.sh b/bin/fm-remote-secondmate-control.sh index 57baadf42f5..9f9e4a6efbf 100755 --- a/bin/fm-remote-secondmate-control.sh +++ b/bin/fm-remote-secondmate-control.sh @@ -397,6 +397,9 @@ cmd_launch() { # Relaunch through the ordinary host-local control plane. The remote mate is # local from this host's perspective, while the explicit profile comes from the # parent because config/secondmate-harness belongs to a different home here. +# The replacement is pinned to the herdr backend exactly like launch, so a +# remote home configured for tmux cannot pull the relaunched mate off the +# endpoint the route depends on. cmd_relaunch() { local id=$1 harness=$2 model=$3 effort=$4 validate_id "$id" @@ -418,7 +421,8 @@ cmd_relaunch() { FM_DATA_OVERRIDE="$CONTROL_DATA" FM_CONFIG_OVERRIDE="$TARGET_HOME/config" \ FM_SKIP_SECONDMATE_INHERIT=1 FM_SKIP_SECONDMATE_SYNC=1 \ "$SCRIPT_DIR/fm-control.sh" "$id" relaunch \ - --harness "$harness" --model "$model" --effort "$effort" + --harness "$harness" --model "$model" --effort "$effort" \ + --backend herdr } cmd_send() { diff --git a/docs/remote-secondmates.md b/docs/remote-secondmates.md index c9a0e674373..bb9dc63a1a1 100644 --- a/docs/remote-secondmates.md +++ b/docs/remote-secondmates.md @@ -183,7 +183,7 @@ The primary resolves both configured secondmate profiles, runs the same readines [`runpod-secondmates.md`](runpod-secondmates.md) owns the provider-specific crew route that converges after this inherited transfer. The remote response supplies the actually launched harness, model, effort, and fallback metadata for the primary record; reusing an already-live endpoint preserves that stored profile instead of relabeling it from a new request. All remote secondmates on one host share `fm-remote` and retain separate `2ndmate-` workspaces inside it. -An explicit request for any other backend is refused rather than honored, and the remote host refuses one too. +An explicit request for any other backend is refused rather than honored, and the remote host refuses one too; a remote relaunch pins the replacement to Herdr like launch does and refuses a recorded endpoint on any other backend. An existing remote endpoint recorded in another Herdr session, including `default`, is classified as unverified and left untouched; launch, liveness recovery, control, and retirement refuse it until an operator explicitly migrates it instead of attempting a live cutover. A launch after a host has drifted out of readiness fails with the doctor's own gap text instead of leaving a half-created endpoint. OMP is accepted for the remote second-mate agent itself, as either the primary or configured fallback harness, through the same verified-adapter boundary as the other supported harnesses. diff --git a/tests/fm-remote-secondmate-lifecycle-e2e.test.sh b/tests/fm-remote-secondmate-lifecycle-e2e.test.sh index 12b24e86386..d5e1397128f 100755 --- a/tests/fm-remote-secondmate-lifecycle-e2e.test.sh +++ b/tests/fm-remote-secondmate-lifecycle-e2e.test.sh @@ -761,6 +761,21 @@ cmp -s "$TMP_ROOT/remote-ios-legacy-before-refusal.meta" "$remote_route_meta" \ assert_present "$TMUX_STATE" "remote refusal killed the alive legacy endpoint" cmp -s "$TMP_ROOT/registry-before-nonherdr.md" "$PARENT/data/secondmates.md" \ || fail "remote legacy refusal removed or changed the registry route" +set +e +remote_env "$ROOT/bin/fm-on.sh" ios fm-remote-secondmate-control.sh relaunch ios codex - - \ + > "$TMP_ROOT/legacy-relaunch-refusal.out" 2>&1 +legacy_relaunch_rc=$? +set -e +[ "$legacy_relaunch_rc" -ne 0 ] || fail "remote control relaunched an alive legacy tmux endpoint" +assert_grep "endpoint is recorded on backend 'tmux', expected 'herdr'" "$TMP_ROOT/legacy-relaunch-refusal.out" \ + "remote relaunch refusal did not name the endpoint's recorded backend" +cmp -s "$TMP_ROOT/remote-ios-legacy-before-refusal.meta" "$remote_route_meta" \ + || fail "remote relaunch refusal changed the legacy endpoint metadata" +cmp -s "$TMP_ROOT/parent-ios-before-nonherdr.meta" "$PARENT/state/ios.meta" \ + || fail "remote relaunch refusal rewrote the parent's endpoint metadata" +cmp -s "$TMP_ROOT/registry-before-nonherdr.md" "$PARENT/data/secondmates.md" \ + || fail "remote relaunch refusal removed or changed the registry route" +assert_present "$TMUX_STATE" "remote relaunch refusal killed the alive legacy endpoint" mv -f "$TMP_ROOT/remote-ios-before-legacy.meta" "$remote_route_meta" rm -f "$TMUX_STATE" pass "non-herdr remote endpoints are refused without changing either route" @@ -1829,6 +1844,52 @@ for pointer_state in absent malformed; do done pass 'Remote failed binds repair absent and malformed pointers without deleting saved sessions' +# --- remote OMP relaunch stays pinned to herdr ------------------------------- +# AC1 regression: a remote home whose own config/backend names tmux must not +# pull a remote second mate's relaunch off herdr - launch pins the backend, and +# relaunch pins it the same way. + +OMP_BROKEN_BACKEND_CFG="$OMP_BROKEN_HOME/config/backend" +if [ -f "$OMP_BROKEN_BACKEND_CFG" ]; then + cp -p "$OMP_BROKEN_BACKEND_CFG" "$TMP_ROOT/omp-broken-backend.prior" + omp_broken_backend_prior=1 +else + omp_broken_backend_prior=0 +fi +mkdir -p "$OMP_BROKEN_HOME/config" +printf 'tmux\n' > "$OMP_BROKEN_BACKEND_CFG" +# Relaunch on a dead endpoint skips graceful exit entirely, so the fixture's +# missing exit semantics do not matter: close the live pane and model the +# session lock the same way the failed-bind case does (dead pid). +OMP_BROKEN_LIVE_PANE=$(sed -n 's/^herdr_pane_id=//p' "$OMP_BROKEN_CONTROL/remote-omp-broken.meta") +[ -n "$OMP_BROKEN_LIVE_PANE" ] || fail "live remote OMP endpoint metadata omitted its Herdr pane identity" +"$REMOTE_ROOT/bin/herdr" pane close "$OMP_BROKEN_LIVE_PANE" +omp_dead_pid=$( { sleep 0.05 & echo "$!"; wait; } 2>/dev/null ) +printf '%s\n' "$omp_dead_pid" > "$OMP_BROKEN_STATE/.lock" +OMP_BROKEN_VERSION=$(bash -c '. "$1/bin/fm-primary-watch-version-lib.sh"; fm_primary_watch_version "$2/.omp/extensions/fm-primary-omp.ts" "$2"' \ + _ "$REMOTE_ROOT" "$OMP_BROKEN_HOME") +printf '%s\n%s\n%s\n%s\n' "$OMP_BROKEN_VERSION" "$omp_dead_pid" "$REMOTE_OMP_BUN" "$REMOTE_OMP_BIN" \ + > "$OMP_BROKEN_STATE/.omp-primary-extension-loaded" +remote_env "$ROOT/bin/fm-on.sh" remote-omp-broken \ + fm-remote-secondmate-control.sh relaunch remote-omp-broken omp - - \ + > "$TMP_ROOT/remote-omp-broken-relaunch.out" 2>&1 \ + || fail "remote OMP relaunch on a tmux-configured home was refused:"$'\n'"$(cat "$TMP_ROOT/remote-omp-broken-relaunch.out")" +assert_grep 'backend=herdr' "$OMP_BROKEN_CONTROL/remote-omp-broken.meta" \ + "the relaunched remote OMP endpoint record does not name the herdr backend" +assert_grep 'herdr_session=fm-remote' "$OMP_BROKEN_CONTROL/remote-omp-broken.meta" \ + "the relaunched remote OMP endpoint record does not name the fm-remote session" +assert_no_grep 'window=firstmate:' "$OMP_BROKEN_CONTROL/remote-omp-broken.meta" \ + "the remote OMP relaunch landed on the home's configured tmux backend instead of herdr" +[ "$(remote_env "$ROOT/bin/fm-on.sh" remote-omp-broken \ + fm-remote-secondmate-control.sh state remote-omp-broken)" = alive ] \ + || fail "remote OMP relaunch on a tmux-configured home did not reach a live endpoint" +if [ "$omp_broken_backend_prior" = 1 ]; then + mv -f "$TMP_ROOT/omp-broken-backend.prior" "$OMP_BROKEN_BACKEND_CFG" +else + rm -f "$OMP_BROKEN_BACKEND_CFG" +fi +pass "a remote OMP relaunch pins the herdr backend even when the remote home configures tmux" + pass "remote OMP primary, fallback, result metadata, pane launch, and existing safety refusals hold end to end" echo "ALL TESTS PASSED" diff --git a/tests/fm-spawn-relaunch-dead-endpoint.test.sh b/tests/fm-spawn-relaunch-dead-endpoint.test.sh index e6dbb732067..eae0140fe6e 100644 --- a/tests/fm-spawn-relaunch-dead-endpoint.test.sh +++ b/tests/fm-spawn-relaunch-dead-endpoint.test.sh @@ -1126,6 +1126,97 @@ test_orca_relaunch_still_refuses() { pass "fm-spawn --relaunch: orca remains refused" } +# --- relaunch --backend pin -------------------------------------------------- + +# make_spawn_stub : a launch-owner stand-in that records its argv and +# brings the recorded tmux endpoint up as a live agent, the way a real relaunch +# publication would leave the world. +make_spawn_stub() { + local case_dir=$1 + cat > "$case_dir/spawn-stub" <<'SH' +#!/usr/bin/env bash +set -u +printf '%s\n' "$*" >> "$FM_TEST_SPAWN_ARGS" +window=$(sed -n 's/^window=//p' "$FM_TEST_META") +ses=${window%%:*} +name=${window#*:} +cwd=$(sed -n 's/^worktree=//p' "$FM_TEST_META") +printf '@9\t%s\t%s\t%s\n' "$name" "$cwd" "claude" > "$FM_FAKE_TMUX_STATE/$ses.windows" +SH + chmod +x "$case_dir/spawn-stub" +} + +test_relaunch_backend_pin_refuses_mismatch() { + local rec id meta_prior + id=$(case_id pin-refuse) + rec=$(make_case pin-refuse "$id" pool) + read_case "$rec" + write_pool_state "$CASE_DIR" "$WT_DIR" "fm-$id" + write_slot_marker "$SLOT_DIR" "$id" "$HOME_DIR" + write_meta "$HOME_DIR/state/$id.meta" "$id" tmux "$WT_DIR" "$PROJ_DIR" claude + create_prior_artifacts "$HOME_DIR/state" "$id" + make_spawn_stub "$CASE_DIR" + export FM_TEST_SPAWN_ARGS="$CASE_DIR/spawn-args.log" + export FM_TEST_META="$HOME_DIR/state/$id.meta" + export FM_RESTART_SPAWN_CMD="$CASE_DIR/spawn-stub" + meta_prior="$CASE_DIR/meta-prior" + cp -p "$HOME_DIR/state/$id.meta" "$meta_prior" + + run_control "$CASE_DIR" "$HOME_DIR" "$id" --backend herdr + unset FM_RESTART_SPAWN_CMD FM_TEST_SPAWN_ARGS FM_TEST_META + + [ "$CONTROL_STATUS" -ne 0 ] || fail "a relaunch pinned to a foreign backend should refuse; got: $CONTROL_OUT" + assert_contains "$CONTROL_OUT" "recorded on backend 'tmux', not the pinned 'herdr'" \ + "the pin refusal did not name the recorded and pinned backends" + cmp -s "$meta_prior" "$HOME_DIR/state/$id.meta" \ + || fail "a refused backend pin rewrote the durable endpoint record" + assert_absent "$HOME_DIR/state/$id.control-relaunch" \ + "a refused backend pin left a relaunch journal behind" + assert_absent "$CASE_DIR/spawn-args.log" \ + "a refused backend pin still invoked the launch owner" + pass "fm-control relaunch --backend: a recorded endpoint on another backend refuses untouched" +} + +test_relaunch_backend_pin_forwards_to_spawn() { + local rec id + id=$(case_id pin-forward) + rec=$(make_case pin-forward "$id" pool) + read_case "$rec" + write_pool_state "$CASE_DIR" "$WT_DIR" "fm-$id" + write_slot_marker "$SLOT_DIR" "$id" "$HOME_DIR" + write_meta "$HOME_DIR/state/$id.meta" "$id" tmux "$WT_DIR" "$PROJ_DIR" claude + create_prior_artifacts "$HOME_DIR/state" "$id" + make_spawn_stub "$CASE_DIR" + export FM_TEST_SPAWN_ARGS="$CASE_DIR/spawn-args.log" + export FM_TEST_META="$HOME_DIR/state/$id.meta" + export FM_RESTART_SPAWN_CMD="$CASE_DIR/spawn-stub" + + run_control "$CASE_DIR" "$HOME_DIR" "$id" --backend tmux + unset FM_RESTART_SPAWN_CMD FM_TEST_SPAWN_ARGS FM_TEST_META + + expect_code 0 "$CONTROL_STATUS" "a relaunch pinned to its recorded backend should proceed; got: $CONTROL_OUT" + assert_contains "$CONTROL_OUT" "relaunched $id" "the pinned relaunch did not report success" + assert_grep "$id --relaunch --harness claude --backend tmux" "$CASE_DIR/spawn-args.log" \ + "the pin was not forwarded to the launch owner: $(cat "$CASE_DIR/spawn-args.log" 2>/dev/null)" + pass "fm-control relaunch --backend: a matching pin is forwarded to the launch owner" +} + +test_backend_flag_refused_off_relaunch() { + local rec id + id=$(case_id pin-verb) + rec=$(make_case pin-verb "$id" pool) + read_case "$rec" + write_meta "$HOME_DIR/state/$id.meta" "$id" tmux "$WT_DIR" "$PROJ_DIR" claude + + CONTROL_OUT=$(spawn_env "$CASE_DIR" "$HOME_DIR" "$id" \ + "$CONTROL" "$id" interrupt --backend tmux 2>&1) + CONTROL_STATUS=$? + [ "$CONTROL_STATUS" -ne 0 ] || fail "--backend was accepted on interrupt" + assert_contains "$CONTROL_OUT" "apply to 'relaunch' only" \ + "interrupt --backend did not refuse as a relaunch-only option" + pass "fm-control --backend: non-relaunch verbs refuse the pin" +} + # --- run --------------------------------------------------------------------- test_tmux_gone_relaunch_recreates_in_worktree @@ -1147,5 +1238,8 @@ test_zellij_present_endpoint_refuses test_cmux_absent_relaunch_recreates test_cmux_present_endpoint_refuses test_orca_relaunch_still_refuses +test_relaunch_backend_pin_refuses_mismatch +test_relaunch_backend_pin_forwards_to_spawn +test_backend_flag_refused_off_relaunch pass "all dead-endpoint relaunch tests" From d7223d16db2050de560e75f27526e1dad13eccac Mon Sep 17 00:00:00 2001 From: dnth Date: Mon, 5 Oct 2026 22:23:06 +0800 Subject: [PATCH 2/2] no-mistakes(document): Clarify backend pin refusal guarantees --- bin/fm-control.sh | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/bin/fm-control.sh b/bin/fm-control.sh index bb32e762557..989bae115e4 100755 --- a/bin/fm-control.sh +++ b/bin/fm-control.sh @@ -50,7 +50,8 @@ # --backend pins the replacement's runtime backend instead of # letting the launch owner resolve it from this home's config; a # task whose recorded endpoint sits on a different backend is -# refused before anything is touched. +# refused before a journal write, agent touch, or replacement +# launch. # --lock-preheld is the supervised-recovery handshake: the caller # (bin/fm-stall-recovery.sh) already holds this task's lifecycle # lock, so fm-control verifies the lock's recorded owner is its