diff --git a/bin/fm-helm-lib.sh b/bin/fm-helm-lib.sh index 15bbf029ea4..705e0b9d97b 100755 --- a/bin/fm-helm-lib.sh +++ b/bin/fm-helm-lib.sh @@ -642,7 +642,27 @@ fm_helm_plan_program() { end end ] end; - (record_entries + missing_entries + deleted_entries) as $all + # An unsupported-repository note is a known, standing fact about a task, + # not a fresh event: debounce it through the same divergence-memory shape + # as every other "already told you" marker in this file (kind + # "unsupported-repo", keyed on task id and the exact repo: value). A record + # entry's own note is silenced once that exact value has already been + # acknowledged; it fires again only when the repo: value itself changes, + # and the marker is cleared once the task no longer needs it. + def repo_of($t): ($record_by_id[$t].repo // ""); + def debounce_unsupported($e): + if $e.phase != "record" then $e + elif ($e.note // "") != "" then + (repo_of($e.task) | tojson | @base64) as $fp + | $e + { + note: (if divergence_matches("unsupported-repo"; $e.task; ""; $fp) then "" else $e.note end), + divergence_ops: (($e.divergence_ops // []) + [{kind: "unsupported-repo", action: "keep", item: "", fp: $fp}]) + } + elif divergence_for("unsupported-repo"; $e.task) != null then + $e + {divergence_ops: (($e.divergence_ops // []) + [{kind: "unsupported-repo", action: "remove", item: "", fp: ""}])} + else $e + end; + (record_entries + missing_entries + deleted_entries | map(debounce_unsupported(.))) as $all | ([$all[] | select(.phase == "error")] + [$all[] | select(.phase != "error")])[] | entry(.) JQ diff --git a/bin/fm-helm-sync.sh b/bin/fm-helm-sync.sh index 987c8ecc09b..a35e6acde01 100755 --- a/bin/fm-helm-sync.sh +++ b/bin/fm-helm-sync.sh @@ -85,6 +85,10 @@ # State files under `.helm-*` retain an acknowledgement fingerprint for each # unresolved field divergence. The sync removes an acknowledgement after that # field no longer diverges, so repeated forced reads do not requeue its wake. +# The same shape debounces the unsupported-repository fallback note (kind +# `unsupported-repo`, keyed on task id and the exact `repo:` value): it prints +# once per task per distinct value instead of on every full replan, and fires +# again only when that task's `repo:` value actually changes. # state/helm-deleted.tsv (mode 0600) retains a captain deletion tombstone while # that task remains in any discovered backlog. It suppresses recreation even # after the captain resolves the hold by marking the task Done. The tombstone @@ -125,6 +129,7 @@ DIVERGENCE_FILES=( "$STATE_PATH/.helm-conflict-body" "$STATE_PATH/.helm-new-card" "$STATE_PATH/.helm-card-deleted" + "$STATE_PATH/.helm-unsupported-repo" ) CARDS_FILE="$STATE_PATH/helm-cards.tsv" DELETED_FILE="$STATE_PATH/helm-deleted.tsv" @@ -803,6 +808,7 @@ while read_entry; do ;; record:skip) [ -z "$tombstone" ] || printf '%s\n' "$tombstone" >>"$NEW_DELETED" + apply_divergence_ops "$divergence_ops" || helm_fail_open "could not update Helm divergence memory" continue ;; record:none|record:create|record:update) diff --git a/bin/fm-helm-watch.sh b/bin/fm-helm-watch.sh index 5f145c36567..f28361b15d0 100755 --- a/bin/fm-helm-watch.sh +++ b/bin/fm-helm-watch.sh @@ -14,6 +14,12 @@ # The underlying sync is the sole board writer and aggregates every local # secondmate backlog, so this main-home check covers their transitions too. # +# A clean run can still print its own informational notes alongside the exact +# success line: an already-known unsupported-repository fallback, or losing +# the lock race to a concurrent sync. Neither means the run needs attention, +# so those exact lines are stripped before the success check below, the same +# way the three silent shapes are matched by exact text. +# # Usage: bin/fm-helm-watch.sh set -u @@ -24,7 +30,11 @@ if ! OUTPUT=$("$SCRIPT_DIR/fm-helm-sync.sh" 2>&1); then exit 0 fi -case "$OUTPUT" in +SIGNIFICANT=$(printf '%s\n' "$OUTPUT" | grep -v \ + -e '^fm-helm-sync: unsupported repository .*; using other$' \ + -e '^fm-helm-sync: another Helm sync is already running$') + +case "$SIGNIFICANT" in ''|'fm-helm-sync: synchronized') exit 0 ;; 'fm-helm-sync: partial: '*' cards remain') exit 0 ;; *) printf '%s\n' "$OUTPUT" ;; diff --git a/docs/configuration.md b/docs/configuration.md index d5f43d1580f..b5f8aeadaf8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -136,6 +136,7 @@ Bootstrap automatically registers `state/helm-sync.check.sh` and `state/helm-boa The watcher runs the sync and authenticated board poll at its ordinary check cadence, and the combined backlog hash avoids a GitHub call when no local home changed. That one main-home writer covers every discovered local secondmate backlog, so secondmates do not need copied Helm configuration or competing sync processes. If a fail-open sync skip would leave a backlog item unsynchronized, its diagnostic becomes a durable watcher `check` wake instead of being silent. +An unsupported `repo:` value maps to the board's `other` bucket and prints one informational fallback note per task and repository value, but that note does not wake the watcher. One run reads the board once, parses every backlog once, and computes the whole reconciliation as one plan before it writes anything, so a fleet-sized backlog plans in well under a second and the run's 25-second budget is spent on card writes. Every landed card write is recorded in `state/helm-cards.tsv` immediately, so a run stopped by its budget, a failed request, or the watcher's check timeout keeps what it already did. diff --git a/tests/fm-helm-sync.test.sh b/tests/fm-helm-sync.test.sh index 7df806f1498..ed90690cb0e 100644 --- a/tests/fm-helm-sync.test.sh +++ b/tests/fm-helm-sync.test.sh @@ -1322,11 +1322,105 @@ EOF board_json '[]' > "$case_dir/board.json" : > "$case_dir/gh.log"; : > "$case_dir/tasks-axi.log" out=$(run_watch "$case_dir" "$fb") || fail "diagnostic watcher adapter exited nonzero: $out" -assert_contains "$out" "unsupported repository unrecognised-project for unsupported-task; using other" \ - "an unsupported backlog project did not emit its fallback diagnostic" +[ -z "$out" ] || fail "an unsupported repository's own fallback note should not wake the watcher: $out" grep -F 'optionId=other-project' "$case_dir/gh.log" >/dev/null \ || fail "an unsupported backlog project was not synced into the other bucket" -pass "watcher adapter retains unsupported projects in the other bucket" +pass "watcher adapter retains unsupported projects in the other bucket without waking on its own note" + +# --------------------------------------------------------------------------- +# Unsupported-repository note: debounced per task per distinct repo: value +# (durable marker), independent of the whole-fleet replan hash. +# --------------------------------------------------------------------------- +case_dir="$TMP_ROOT/unsupported-repo-debounce" +mkdir -p "$case_dir/home/config" "$case_dir/home/data" "$case_dir/home/state" +fb=$(install_fakes "$case_dir") +printf '{"owner":"geojitsu","number":2}\n' > "$case_dir/home/config/helm.json" +cat > "$case_dir/home/data/backlog.md" <<'EOF' +# Backlog + +## Queued +- [ ] debounce-task - Must not disappear (repo: repo-prefix repo-suffix) (kind: ship) (since: 2026-09-09) +## Done +EOF +board_json '[]' > "$case_dir/board.json" +: > "$case_dir/gh.log"; : > "$case_dir/tasks-axi.log" + +out=$(run_sync "$case_dir" "$fb" 2>&1) || fail "unsupported-repo debounce seed run exited nonzero: $out" +assert_contains "$out" $'unsupported repository repo-prefix\trepo-suffix for debounce-task; using other' \ + "the first sync of an unsupported repository did not emit its diagnostic" +repo_fp=$(printf '%s' $'repo-prefix\trepo-suffix' | jq -Rsc -r 'tojson | @base64') +grep -Fq $'debounce-task\t\t'"$repo_fp" "$case_dir/home/state/.helm-unsupported-repo" \ + || fail "the durable marker did not safely encode the task's repo: value: $(cat "$case_dir/home/state/.helm-unsupported-repo" 2>&1)" +mv "$case_dir/board-state.json" "$case_dir/board.json" +pass "an unsupported repository's first sync emits its diagnostic and leaves a durable marker" + +# An unrelated backlog change invalidates the whole-fleet debounce hash and +# forces a full replan; the already-known unsupported task must stay silent. +cat > "$case_dir/home/data/backlog.md" <<'EOF' +# Backlog + +## Queued +- [ ] debounce-task - Must not disappear (repo: repo-prefix repo-suffix) (kind: ship) (since: 2026-09-09) +- [ ] unrelated-task - Elsewhere entirely (repo: firstmate) (kind: ship) (since: 2026-09-10) +## Done +EOF +out=$(run_sync "$case_dir" "$fb" 2>&1) || fail "unrelated-task replan exited nonzero: $out" +assert_not_contains "$out" "unsupported repository" \ + "an unrelated fleet replan re-fired an already-known unsupported-repository note: $out" +grep -F 'addProjectV2DraftIssue' "$case_dir/gh.log" >/dev/null \ + || fail "the unrelated task's own card was not created by the same replan" +mv "$case_dir/board-state.json" "$case_dir/board.json" +pass "an unrelated fleet replan does not re-fire an already-known unsupported-repository note" + +# Changing the task's own repo: value is a new fact and must fire again. +cat > "$case_dir/home/data/backlog.md" <<'EOF' +# Backlog + +## Queued +- [ ] debounce-task - Must not disappear (repo: another-unrecognised-project) (kind: ship) (since: 2026-09-09) +- [ ] unrelated-task - Elsewhere entirely (repo: firstmate) (kind: ship) (since: 2026-09-10) +## Done +EOF +out=$(run_sync "$case_dir" "$fb" 2>&1) || fail "changed repo: value run exited nonzero: $out" +assert_contains "$out" "unsupported repository another-unrecognised-project for debounce-task; using other" \ + "a task's repo: value changing to a new unsupported value did not re-fire the note" +repo_fp=$(printf '%s' another-unrecognised-project | jq -Rsc -r 'tojson | @base64') +grep -Fq $'debounce-task\t\t'"$repo_fp" "$case_dir/home/state/.helm-unsupported-repo" \ + || fail "the durable marker did not move to the task's new repo: value" +pass "a task's repo: value changing to a new unsupported value re-fires the note" + +# --------------------------------------------------------------------------- +# Losing the lock race is a benign fail-open on its own and must not wake. +# --------------------------------------------------------------------------- +case_dir="$TMP_ROOT/watcher-lock-contention" +mkdir -p "$case_dir/home/config" "$case_dir/home/data" "$case_dir/home/state" +fb=$(install_fakes "$case_dir") +printf '{"owner":"geojitsu","number":2}\n' > "$case_dir/home/config/helm.json" +cat > "$case_dir/home/data/backlog.md" <<'EOF' +# Backlog + +## Queued +- [ ] lock-task - Reaches the board while another sync holds the lock (repo: firstmate) (kind: ship) (since: 2026-09-09) +## Done +EOF +board_json '[]' > "$case_dir/board.json" +: > "$case_dir/gh.log"; : > "$case_dir/tasks-axi.log" +out=$( + FM_HOME="$case_dir/home" + FM_ROOT_OVERRIDE="$ROOT" + # shellcheck source=bin/fm-wake-lib.sh + . "$ROOT/bin/fm-wake-lib.sh" + fm_lock_try_acquire "$case_dir/home/state/.helm-sync.lock" || exit 3 + run_watch "$case_dir" "$fb" + status=$? + fm_lock_release "$case_dir/home/state/.helm-sync.lock" || true + exit "$status" +) +rc=$? +[ "$rc" -ne 3 ] || fail "test setup could not hold the Helm sync lock to simulate contention" +[ "$rc" -eq 0 ] || fail "watcher adapter exited nonzero during lock contention: $out" +[ -z "$out" ] || fail "losing the lock race to a concurrent sync should stay silent: $out" +pass "the watcher stays silent when a run only loses the lock race to a concurrent sync" for mode in no-config noauth scope network; do cd_dir="$TMP_ROOT/fail-$mode"