From 17e8152b0ffe460781e1c904e2285abb2cb45445 Mon Sep 17 00:00:00 2001 From: firstmate-crewmate Date: Thu, 17 Sep 2026 00:29:55 +0000 Subject: [PATCH 1/2] fix(bin): stop Helm sync's per-board schema check from exceeding argv limits fm-helm-sync.sh passed a board group's whole desired-records JSON as a literal jq --argjson command-line argument. Real fleet card bodies exceed Linux's per-argument limit, so the schema check silently failed on every board at real fleet scale, reporting fields and options as unavailable even when they were present. Switch to --slurpfile, the same pattern the file's plan-building jq call already uses. Audited fm-helm-lib.sh and fm-helm-project-map.sh for the same pattern; every other --argjson call in the Helm sync path passes a small, bounded value, so nothing else needed changing. Adds a regression test with a fleet-scale (100-card) fixture whose combined desired-records JSON exceeds the real argv limit. --- bin/fm-helm-sync.sh | 5 ++- ...09-17-helm-sync-schema-check-argv-limit.md | 26 +++++++++++ tests/fm-helm-sync.test.sh | 44 +++++++++++++++++++ 3 files changed, 73 insertions(+), 2 deletions(-) create mode 100644 docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md diff --git a/bin/fm-helm-sync.sh b/bin/fm-helm-sync.sh index 7a2ac80e12c..9dbc0a2f85f 100755 --- a/bin/fm-helm-sync.sh +++ b/bin/fm-helm-sync.sh @@ -983,8 +983,9 @@ process_board() { # Project field may not have that option yet) so the per-board hot path # stays a single jq call, same as before this check grew a second part. schema_ok=false missing_projects= - IFS=$'\t' read -r schema_ok missing_projects < <(jq -r --argjson records "$(cat "$group_desired")" --argjson registered "$REGISTERED_PROJECTS" --slurpfile routing "$ROUTING_JSON" --arg owner "$owner" --argjson number "$number" --arg default_owner "$OWNER" --argjson default_number "$PROJECT_NUMBER" --arg dispatch "$DISPATCH_STATUS" ' - .data.user.projectV2.fields.nodes as $fields + IFS=$'\t' read -r schema_ok missing_projects < <(jq -r --slurpfile records_wrap "$group_desired" --argjson registered "$REGISTERED_PROJECTS" --slurpfile routing "$ROUTING_JSON" --arg owner "$owner" --argjson number "$number" --arg default_owner "$OWNER" --argjson default_number "$PROJECT_NUMBER" --arg dispatch "$DISPATCH_STATUS" ' + ($records_wrap[0]) as $records + | .data.user.projectV2.fields.nodes as $fields | (def has_option($field; $name): any($fields[]; .name == $field and .__typename == "ProjectV2SingleSelectField" and any(.options[]?; .name == $name)); any($fields[]; .name == "Status" and .__typename == "ProjectV2SingleSelectField") and any($fields[]; .name == "Project" and .__typename == "ProjectV2SingleSelectField") diff --git a/docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md b/docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md new file mode 100644 index 00000000000..27ced694aac --- /dev/null +++ b/docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md @@ -0,0 +1,26 @@ +--- +title: Helm board sync failed its schema check on real fleet scale +description: jq rejected an unbounded desired-records array passed through argv, silently failing every board's schema check once a fleet's full card bodies grew past the per-argument limit. +--- + +## Symptom + +`bin/fm-helm-sync.sh` reported every board "could not be reconciled" with "required Helm fields or options are unavailable", even though the board's Status, Project, Kind, and Priority fields and options were all present and correct. +The failure was silent about its real cause: the schema-check jq call's stderr is intentionally suppressed elsewhere for legitimate per-board diagnostics, so the actual `jq: Argument list too long` never surfaced. +Small local test fixtures never reproduced it; the real fleet's ~84+ live tasks with full card bodies did. + +## Root cause + +`process_board()` built the per-board schema check with `jq -r --argjson records "$(cat "$group_desired")" ...`, passing the whole board group's desired-records JSON as a single command-line argument. +Linux applies a 128 KiB maximum to an individual argument, independently of the total process argument limit - the same class of failure as [2026-09-05-fleet-snapshot-argv-limit](2026-09-05-fleet-snapshot-argv-limit.md). +A single real card's full body, let alone a whole board group's array of them, routinely exceeds that. + +## Fix + +The schema check now reads `$group_desired` with `--slurpfile records_wrap`, then unwraps it as `($records_wrap[0]) as $records` at the top of the jq program, matching the pattern the same file's plan-building call (`fm_helm_plan_program`, a few lines below) already used for the same file. +`bin/fm-helm-project-map.sh` and `bin/fm-helm-lib.sh` were audited for the same unsafe `--argjson "$(...)"` pattern; every other `--argjson` call in Helm's sync path passes a small, bounded value (project-name lists, single GraphQL item responses, and similar) well under the per-argument limit, so no other call needed changing. + +## Prevention + +`tests/fm-helm-sync.test.sh` ("the per-board schema check survives a desired-records array larger than ARG_MAX") builds a 100-card fleet whose combined desired-records JSON exceeds the 2 MiB total argv limit, with every card already matching its board twin so the run needs no writes and isolates the read path. +It asserts the backlog fixture is actually larger than that limit before trusting the run, and that the sync still reports `synchronized`. diff --git a/tests/fm-helm-sync.test.sh b/tests/fm-helm-sync.test.sh index 113c3bf2074..b79b5190fdc 100644 --- a/tests/fm-helm-sync.test.sh +++ b/tests/fm-helm-sync.test.sh @@ -1590,6 +1590,50 @@ rc=$? [ -z "$out" ] || fail "losing the lock race to a concurrent sync should stay silent: $out" pass "the watcher stays silent when a run only loses the lock race to a concurrent sync" +# --------------------------------------------------------------------------- +# Fleet-scale schema check: a real fleet's full card bodies push a board +# group's combined desired-records JSON well past the OS argv limit +# (ARG_MAX, 2097152 bytes on this host). The per-board schema check must +# read that array through a file, never as a literal jq argument, or it +# fails closed on every board every time - see fm-helm-sync-schema-check- +# argmax-001. Every card already matches its board twin so this run needs +# no writes; the only thing under test is whether the read path survives +# an oversized array, not per-row write throughput (that is covered above). +# --------------------------------------------------------------------------- +case_dir="$TMP_ROOT/argmax-scale" +mkdir -p "$case_dir/home/config" "$case_dir/home/data" "$case_dir/home/state" +fb=$(install_fakes "$case_dir") +printf '{"owner":"fixture-owner","number":999}\n' > "$case_dir/home/config/helm.json" + +argmax_task_count=100 +argmax_note_pad=$(printf 'x%.0s' $(seq 1 25000)) + +{ + printf '%s\n\n' "# Backlog" + printf '%s\n' "## Queued" + for i in $(seq 1 "$argmax_task_count"); do + printf -- '- [ ] argmax-task-%d - Argmax task %d (repo: fixture-firstmate) (kind: ship) (priority: 3) (since: 2026-09-02)\n' "$i" "$i" + printf ' note for task %d %s\n' "$i" "$argmax_note_pad" + done + printf '%s\n' "## Done" +} > "$case_dir/home/data/backlog.md" +[ "$(wc -c < "$case_dir/home/data/backlog.md")" -gt 2097152 ] \ + || fail "argmax backlog fixture is not larger than ARG_MAX (2097152 bytes); strengthen the padding before trusting this test to catch an argv-limit regression" + +argmax_items="$case_dir/argmax-items.jsonl" +: > "$argmax_items" +for i in $(seq 1 "$argmax_task_count"); do + scale_item "argmax-task-$i" "Argmax task $i" "$(scale_body "argmax-task-$i" P3 2026-09-02 "note for task $i $argmax_note_pad")" Queued queued-status P3 p3-priority >> "$argmax_items" +done +board_json "$(jq -s '.' "$argmax_items")" > "$case_dir/board.json" + +: > "$case_dir/gh.log"; : > "$case_dir/tasks-axi.log" +out=$(run_sync "$case_dir" "$fb" 2>&1) || fail "argmax-scale sync exited nonzero: $out" +assert_contains "$out" "fm-helm-sync: synchronized" "a run against an oversized desired-records array did not finish: $out" +[ "$(grep -c 'query=mutation(' "$case_dir/gh.log")" -eq 0 ] \ + || fail "an already-matching fleet-scale board should need no writes: $(grep -F 'query=mutation(' "$case_dir/gh.log" | head -3)" +pass "the per-board schema check survives a desired-records array larger than ARG_MAX" + for mode in no-config noauth scope network; do cd_dir="$TMP_ROOT/fail-$mode" mkdir -p "$cd_dir/home/config" "$cd_dir/home/data" "$cd_dir/home/state" From 02283e0bd28fa98e32e7d155069c5faea40a7370 Mon Sep 17 00:00:00 2001 From: firstmate-crewmate Date: Thu, 17 Sep 2026 01:35:25 +0000 Subject: [PATCH 2/2] no-mistakes(document): Register new Helm sync bug doc in the docs audience inventory --- docs/documentation-audiences.json | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/documentation-audiences.json b/docs/documentation-audiences.json index d8e46ae713c..750a95b6e8f 100644 --- a/docs/documentation-audiences.json +++ b/docs/documentation-audiences.json @@ -512,6 +512,10 @@ "path": "docs/bugs/2026-09-15-backlog-close-non-github-pr-link.md", "audience": "maintainer-architecture" }, + { + "path": "docs/bugs/2026-09-17-helm-sync-schema-check-argv-limit.md", + "audience": "maintainer-architecture" + }, { "path": "docs/calm-mode-feasibility.md", "audience": "maintainer-verification"