diff --git a/bin/fm-helm-sync.sh b/bin/fm-helm-sync.sh index 7a2ac80e12c..9dbc0a2f85f 100755 --- a/bin/fm-helm-sync.sh +++ b/bin/fm-helm-sync.sh @@ -983,8 +983,9 @@ process_board() { # Project field may not have that option yet) so the per-board hot path # stays a single jq call, same as before this check grew a second part. schema_ok=false missing_projects= - IFS=$'\t' read -r schema_ok missing_projects < <(jq -r --argjson records "$(cat "$group_desired")" --argjson registered "$REGISTERED_PROJECTS" --slurpfile routing "$ROUTING_JSON" --arg owner "$owner" --argjson number "$number" --arg default_owner "$OWNER" --argjson default_number "$PROJECT_NUMBER" --arg dispatch "$DISPATCH_STATUS" ' - .data.user.projectV2.fields.nodes as $fields + IFS=$'\t' read -r schema_ok missing_projects < <(jq -r --slurpfile records_wrap "$group_desired" --argjson registered "$REGISTERED_PROJECTS" --slurpfile routing "$ROUTING_JSON" --arg owner "$owner" --argjson number "$number" --arg default_owner "$OWNER" --argjson default_number "$PROJECT_NUMBER" --arg dispatch "$DISPATCH_STATUS" ' + ($records_wrap[0]) as $records + | .data.user.projectV2.fields.nodes as $fields | (def has_option($field; $name): any($fields[]; .name == $field and .__typename == "ProjectV2SingleSelectField" and any(.options[]?; .name == $name)); any($fields[]; .name == "Status" and .__typename == "ProjectV2SingleSelectField") and any($fields[]; .name == "Project" and .__typename == "ProjectV2SingleSelectField") diff --git a/docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md b/docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md new file mode 100644 index 00000000000..27ced694aac --- /dev/null +++ b/docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md @@ -0,0 +1,26 @@ +--- +title: Helm board sync failed its schema check on real fleet scale +description: jq rejected an unbounded desired-records array passed through argv, silently failing every board's schema check once a fleet's full card bodies grew past the per-argument limit. +--- + +## Symptom + +`bin/fm-helm-sync.sh` reported every board "could not be reconciled" with "required Helm fields or options are unavailable", even though the board's Status, Project, Kind, and Priority fields and options were all present and correct. +The failure was silent about its real cause: the schema-check jq call's stderr is intentionally suppressed elsewhere for legitimate per-board diagnostics, so the actual `jq: Argument list too long` never surfaced. +Small local test fixtures never reproduced it; the real fleet's ~84+ live tasks with full card bodies did. + +## Root cause + +`process_board()` built the per-board schema check with `jq -r --argjson records "$(cat "$group_desired")" ...`, passing the whole board group's desired-records JSON as a single command-line argument. +Linux applies a 128 KiB maximum to an individual argument, independently of the total process argument limit - the same class of failure as [2026-09-05-fleet-snapshot-argv-limit](2026-09-05-fleet-snapshot-argv-limit.md). +A single real card's full body, let alone a whole board group's array of them, routinely exceeds that. + +## Fix + +The schema check now reads `$group_desired` with `--slurpfile records_wrap`, then unwraps it as `($records_wrap[0]) as $records` at the top of the jq program, matching the pattern the same file's plan-building call (`fm_helm_plan_program`, a few lines below) already used for the same file. +`bin/fm-helm-project-map.sh` and `bin/fm-helm-lib.sh` were audited for the same unsafe `--argjson "$(...)"` pattern; every other `--argjson` call in Helm's sync path passes a small, bounded value (project-name lists, single GraphQL item responses, and similar) well under the per-argument limit, so no other call needed changing. + +## Prevention + +`tests/fm-helm-sync.test.sh` ("the per-board schema check survives a desired-records array larger than ARG_MAX") builds a 100-card fleet whose combined desired-records JSON exceeds the 2 MiB total argv limit, with every card already matching its board twin so the run needs no writes and isolates the read path. +It asserts the backlog fixture is actually larger than that limit before trusting the run, and that the sync still reports `synchronized`. diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index d8e46ae713c..750a95b6e8f 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -512,6 +512,10 @@ "path": "docs/bugs/2026-09-15-backlog-close-non-github-pr-link.md", "audience": "maintainer-architecture" }, + { + "path": "docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md", + "audience": "maintainer-architecture" + }, { "path": "docs/calm-mode-feasibility.md", "audience": "maintainer-verification" diff --git a/tests/fm-helm-sync.test.sh b/tests/fm-helm-sync.test.sh index 113c3bf2074..b79b5190fdc 100644 --- a/tests/fm-helm-sync.test.sh +++ b/tests/fm-helm-sync.test.sh @@ -1590,6 +1590,50 @@ rc=$? [ -z "$out" ] || fail "losing the lock race to a concurrent sync should stay silent: $out" pass "the watcher stays silent when a run only loses the lock race to a concurrent sync" +# --------------------------------------------------------------------------- +# Fleet-scale schema check: a real fleet's full card bodies push a board +# group's combined desired-records JSON well past the OS argv limit +# (ARG_MAX, 2097152 bytes on this host). The per-board schema check must +# read that array through a file, never as a literal jq argument, or it +# fails closed on every board every time - see fm-helm-sync-schema-check- +# argmax-001. Every card already matches its board twin so this run needs +# no writes; the only thing under test is whether the read path survives +# an oversized array, not per-row write throughput (that is covered above). +# --------------------------------------------------------------------------- +case_dir="$TMP_ROOT/argmax-scale" +mkdir -p "$case_dir/home/config" "$case_dir/home/data" "$case_dir/home/state" +fb=$(install_fakes "$case_dir") +printf '{"owner":"fixture-owner","number":999}\n' > "$case_dir/home/config/helm.json" + +argmax_task_count=100 +argmax_note_pad=$(printf 'x%.0s' $(seq 1 25000)) + +{ + printf '%s\n\n' "# Backlog" + printf '%s\n' "## Queued" + for i in $(seq 1 "$argmax_task_count"); do + printf -- '- [ ] argmax-task-%d - Argmax task %d (repo: fixture-firstmate) (kind: ship) (priority: 3) (since: 2026-09-02)\n' "$i" "$i" + printf ' note for task %d %s\n' "$i" "$argmax_note_pad" + done + printf '%s\n' "## Done" +} > "$case_dir/home/data/backlog.md" +[ "$(wc -c < "$case_dir/home/data/backlog.md")" -gt 2097152 ] \ + || fail "argmax backlog fixture is not larger than ARG_MAX (2097152 bytes); strengthen the padding before trusting this test to catch an argv-limit regression" + +argmax_items="$case_dir/argmax-items.jsonl" +: > "$argmax_items" +for i in $(seq 1 "$argmax_task_count"); do + scale_item "argmax-task-$i" "Argmax task $i" "$(scale_body "argmax-task-$i" P3 2026-09-02 "note for task $i $argmax_note_pad")" Queued queued-status P3 p3-priority >> "$argmax_items" +done +board_json "$(jq -s '.' "$argmax_items")" > "$case_dir/board.json" + +: > "$case_dir/gh.log"; : > "$case_dir/tasks-axi.log" +out=$(run_sync "$case_dir" "$fb" 2>&1) || fail "argmax-scale sync exited nonzero: $out" +assert_contains "$out" "fm-helm-sync: synchronized" "a run against an oversized desired-records array did not finish: $out" +[ "$(grep -c 'query=mutation(' "$case_dir/gh.log")" -eq 0 ] \ + || fail "an already-matching fleet-scale board should need no writes: $(grep -F 'query=mutation(' "$case_dir/gh.log" | head -3)" +pass "the per-board schema check survives a desired-records array larger than ARG_MAX" + for mode in no-config noauth scope network; do cd_dir="$TMP_ROOT/fail-$mode" mkdir -p "$cd_dir/home/config" "$cd_dir/home/data" "$cd_dir/home/state"