diff --git a/bin/fm-claude-automic-vault-launch.sh b/bin/fm-claude-automic-vault-launch.sh index 632e19262d7..8a7134b2df9 100755 --- a/bin/fm-claude-automic-vault-launch.sh +++ b/bin/fm-claude-automic-vault-launch.sh @@ -37,8 +37,23 @@ if [ "${1:-}" = --injected ]; then worker_path=${FM_CLAUDE_AV_WORKER_PATH:-/usr/bin:/bin:/usr/sbin:/sbin} unset FM_CLAUDE_AV_WORKER_PATH worker_args=() + drop_settings_value=0 + # Drop any --settings the worker template carries so the credential-scrubbing + # AV settings ($settings, exec'd below) is the sole --settings Claude sees: + # a duplicate could win last and silently drop the apiKeyHelper/auth-env + # neutralization, breaking the "no fallback credential" guarantee in + # fm-claude-automic-vault-lib.sh. That AV settings also carries feedbackDrafts:off. for worker_arg in "$@"; do - [ "$worker_arg" = --dangerously-skip-permissions ] || worker_args+=("$worker_arg") + if [ "$drop_settings_value" = 1 ]; then + drop_settings_value=0 + continue + fi + case "$worker_arg" in + --dangerously-skip-permissions) continue ;; + --settings) drop_settings_value=1; continue ;; + --settings=*) continue ;; + esac + worker_args+=("$worker_arg") done : > "$ready" || exit 1 exec 1>&3 2>&4 3>&- 4>&- diff --git a/bin/fm-claude-automic-vault-lib.sh b/bin/fm-claude-automic-vault-lib.sh index f4324aad220..0445cffb52e 100644 --- a/bin/fm-claude-automic-vault-lib.sh +++ b/bin/fm-claude-automic-vault-lib.sh @@ -37,7 +37,7 @@ FM_CLAUDE_AV_TIMEOUT=${FM_CLAUDE_AV_TIMEOUT:-45} FM_CLAUDE_AV_RELEASE_BASE=https://downloads.claude.ai/claude-code-releases FM_CLAUDE_AV_MANIFEST_CHECKSUMS="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/fm-claude-automic-vault-manifests.sha256" FM_CLAUDE_AV_QUALIFIED_VERSIONS="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/fm-claude-automic-vault-qualified-versions" -FM_CLAUDE_AV_SETTINGS='{"apiKeyHelper":null,"env":{"ANTHROPIC_API_KEY":null,"ANTHROPIC_AUTH_TOKEN":null,"ANTHROPIC_BASE_URL":null,"ANTHROPIC_BEDROCK_BASE_URL":null,"ANTHROPIC_VERTEX_BASE_URL":null,"ANTHROPIC_FOUNDRY_BASE_URL":null,"CLAUDE_CODE_USE_BEDROCK":null,"CLAUDE_CODE_USE_VERTEX":null,"CLAUDE_CODE_USE_FOUNDRY":null,"AWS_BEARER_TOKEN_BEDROCK":null}}' +FM_CLAUDE_AV_SETTINGS='{"apiKeyHelper":null,"feedbackDrafts":"off","env":{"ANTHROPIC_API_KEY":null,"ANTHROPIC_AUTH_TOKEN":null,"ANTHROPIC_BASE_URL":null,"ANTHROPIC_BEDROCK_BASE_URL":null,"ANTHROPIC_VERTEX_BASE_URL":null,"ANTHROPIC_FOUNDRY_BASE_URL":null,"CLAUDE_CODE_USE_BEDROCK":null,"CLAUDE_CODE_USE_VERTEX":null,"CLAUDE_CODE_USE_FOUNDRY":null,"AWS_BEARER_TOKEN_BEDROCK":null}}' FM_CLAUDE_AV_ERROR= FM_CLAUDE_AV_BIN= FM_CLAUDE_BIN= diff --git a/docs/herdr-backend.md b/docs/herdr-backend.md index b47f2512e76..a734f89ad77 100644 --- a/docs/herdr-backend.md +++ b/docs/herdr-backend.md @@ -290,7 +290,7 @@ Capability matching checks the bounded schema in-process, avoiding the early-exi Polling runs every cycle and remains the permanent fallback when protocol 16, the event schema, Python, connection, subscription, or repeated reader execution is unavailable. There is still one watcher process; the event reader is a bounded child of that watcher. -`tests/fm-backend-herdr-eventwait-smoke.test.sh`, `tests/fm-transition-lib.test.sh`, and `tests/fm-supervision-events.test.sh` cover capability, subscribe-then-reconcile ordering, dedupe, exemptions, and polling fallback. +`tests/fm-backend-herdr.test.sh`, `tests/fm-backend-herdr-eventwait-smoke.test.sh`, `tests/fm-transition-lib.test.sh`, and `tests/fm-supervision-events.test.sh` cover capability, subscribe-then-reconcile ordering, dedupe, exemptions, broken-pipe regressions, and polling fallback. ## Away-mode supervisor support diff --git a/docs/verification/claude-automic-vault.md b/docs/verification/claude-automic-vault.md index eb589a2c2d6..a97e762c293 100644 --- a/docs/verification/claude-automic-vault.md +++ b/docs/verification/claude-automic-vault.md @@ -48,6 +48,7 @@ tests/fm-claude-automic-vault.test.sh The test uses fake `av` and fake `claude` executables with synthetic secret bytes held only in process environment. It exercises public provisioning, one-time enable recovery, renewal, and preflight, enabled and disabled spawn behavior, direct, assignment-prefixed, option-prefixed, split-string `env`, literal shell-payload, literal-eval, and classifier-unavailable opted-in raw Claude refusal before endpoint creation, Claude versus non-Claude isolation, missing Vault state, Secret Gate denial, a missing secret, revoked-token rejection, unsupported tool surfaces, inconclusive authentication, executable recursion refusal, model and effort argument preservation, redacted output, persistent secondmate launch and relaunch, inherited opt-in, and a nested worker launched from the inherited home. It executes the captured enabled worker launch and proves the fake Claude process received the injected environment while higher-precedence auth inputs were absent. +It also proves the final Claude exec receives exactly one `--settings` value, that this authoritative value disables `apiKeyHelper` and matching authentication environment settings while preserving `feedbackDrafts: off`, and that no worker-template settings survive into the exec arguments. It then scans every fixture file, captured launch command, fake argv log, and command output to prove the synthetic secret bytes were not persisted or displayed. The fake Claude coverage exercises general launch mechanics but never qualifies or admits a real Claude version; only the actual-candidate command in the qualification section can supply that evidence. diff --git a/tests/fm-backend-cmux.test.sh b/tests/fm-backend-cmux.test.sh index 16875a95cd1..7619ea49b77 100755 --- a/tests/fm-backend-cmux.test.sh +++ b/tests/fm-backend-cmux.test.sh @@ -511,6 +511,29 @@ test_target_ready_fails_when_target_absent() { pass "fm_backend_cmux_target_ready: fails when the workspace/surface is not found (list-panes structural check)" } +test_surface_exists_fails_when_list_panes_call_fails() { + # Regression: a failed `list-panes` call must read as "unknown/not live", never as + # a confirmed surface. Without pipefail the old pipe trusted jq's verdict on whatever + # bytes reached it, so a nonzero list-panes that still printed a matching pane read as + # a false positive. The guard checks the CLI exit and non-empty output before jq. + local dir fb status + dir="$TMP_ROOT/surface-list-panes-fail"; mkdir -p "$dir/fakebin" + cat > "$dir/fakebin/cmux" <<'SH' +#!/usr/bin/env bash +set -u +# Emit a matching-surface payload but fail the call (server error mid-stream). +printf '{"panes":[{"selected_surface_id":"bbbbbbbb-1111-1111-1111-111111111111","surface_ids":["bbbbbbbb-1111-1111-1111-111111111111"]}]}' +exit 1 +SH + chmod +x "$dir/fakebin/cmux" + fb="$dir/fakebin" + PATH="$fb:$PATH" \ + bash -c '. "$0/bin/backends/cmux.sh"; fm_backend_cmux_surface_exists "aaaaaaaa-0000-0000-0000-000000000000" "bbbbbbbb-1111-1111-1111-111111111111"' "$ROOT" + status=$? + [ "$status" -ne 0 ] || fail "surface_exists must not report a surface present when the list-panes call failed" + pass "fm_backend_cmux_surface_exists: a failed list-panes call is not a confirmed surface" +} + test_target_ready_checks_expected_label() { local dir fb title dir="$TMP_ROOT/ready-label-ok"; mkdir -p "$dir/responses" @@ -1131,6 +1154,7 @@ test_ensure_running_fails_fast_on_unauth_without_launching test_create_task_refuses_duplicate_label test_create_task_creates_and_parses_ids test_target_ready_fails_when_target_absent +test_surface_exists_fails_when_list_panes_call_fails test_target_ready_checks_expected_label test_target_ready_rejects_label_mismatch test_capture_trims_locally diff --git a/tests/fm-backend-herdr.test.sh b/tests/fm-backend-herdr.test.sh index 95f4fe50097..68578514d80 100755 --- a/tests/fm-backend-herdr.test.sh +++ b/tests/fm-backend-herdr.test.sh @@ -4541,24 +4541,6 @@ test_events_capable_rejects_schema_missing_a_needle() { pass "fm_backend_herdr_events_capable: schema missing an event needle fails closed" } -test_events_capable_does_not_pipe_schema_through_grep_q() { - # Source-shape lock: the ~220KB schema must not be fed to an early-exit - # consumer via a pipe. That pattern is the broken-pipe defect; reintroducing - # it would re-pollute every watcher cycle. Strip comments so the lock looks - # only at executable lines. - local body code - body=$(sed -n '/^fm_backend_herdr_events_capable()/,/^}/p' "$ROOT/bin/backends/herdr.sh") - [ -n "$body" ] || fail "could not extract fm_backend_herdr_events_capable from herdr.sh" - code=$(printf '%s\n' "$body" | sed -e 's/[[:space:]]*#.*//' -e '/^[[:space:]]*$/d') - if printf '%s\n' "$code" | grep -E '(^|[^[:alnum:]_])grep[[:space:]]' >/dev/null; then - fail "events_capable must not invoke grep on the schema (broken-pipe regression)" - fi - if printf '%s\n' "$code" | grep -E 'printf.*\|' >/dev/null; then - fail "events_capable must not pipe printf of the schema into a consumer (broken-pipe regression)" - fi - pass "fm_backend_herdr_events_capable: source does not pipe schema through early-exit grep -q" -} - # shellcheck source=bin/fm-backend.sh . "$ROOT/bin/fm-backend.sh" @@ -4746,4 +4728,3 @@ test_wait_transition_bad_ack_returns_2_and_cleans_up test_wait_transition_clean_timeout_returns_1 test_events_capable_accepts_large_schema_without_broken_pipe test_events_capable_rejects_schema_missing_a_needle -test_events_capable_does_not_pipe_schema_through_grep_q diff --git a/tests/fm-claude-automic-vault.test.sh b/tests/fm-claude-automic-vault.test.sh index b42530930a1..2fdd27bd5be 100755 --- a/tests/fm-claude-automic-vault.test.sh +++ b/tests/fm-claude-automic-vault.test.sh @@ -20,6 +20,7 @@ ln -s "$ROOT/tests/fixtures/fm-claude-automic-vault-lib.sh" \ AUTH="$TEST_BIN/fm-claude-automic-vault.sh" SPAWN="$TEST_BIN/fm-spawn.sh" REAL_SPAWN="$ROOT/bin/fm-spawn.sh" +AV_LAUNCH="$TEST_BIN/fm-claude-automic-vault-launch.sh" export FM_CLAUDE_AV_TEST_PRODUCTION_ROOT=$ROOT JQ_BIN=$(command -v jq) || fail "test needs jq" NODE_BIN=$(command -v node) || fail "test needs node" @@ -553,7 +554,7 @@ test_provision_recovery_renewal_preflight_and_redaction() { } test_enabled_disabled_and_non_claude_launches() { - local dir home fakebin state record proj wt launchlog output status launch before executed raw id before_claude before_endpoint raw_heredoc tasktmp gotmp raw_index=0 + local dir home fakebin state record proj wt launchlog output status launch before executed raw id before_claude before_endpoint raw_heredoc tasktmp gotmp injected_argv injected_settings injected_worker_args direct_settings settings_count ready raw_index=0 dir="$TMP_ROOT/launches" home="$dir/home" fakebin=$(make_fake_tools "$dir") @@ -592,6 +593,50 @@ test_enabled_disabled_and_non_claude_launches() { "enabled Claude launch required an approval prompt" assert_no_grep 'conflict=' "$state/claude-env.log" "higher-precedence auth environment reached Claude" + injected_argv=$(grep '^claude ' "$state/claude-argv.log" | tail -1) + settings_count=$(printf '%s\n' "$injected_argv" | grep -oE '<--settings(>|=)' | wc -l | tr -d ' ') + [ "$settings_count" = 1 ] \ + || fail "injected Claude exec carried $settings_count --settings flags, expected exactly one authoritative --settings" + injected_settings=$(printf '%s\n' "$injected_argv" | sed -n 's/.*<--settings> <\([^>]*\)>.*/\1/p') + printf '%s' "$injected_settings" | "$JQ_BIN" -e \ + 'has("apiKeyHelper") and .apiKeyHelper == null and .feedbackDrafts == "off" and + (.env | keys | sort) == ([ + "ANTHROPIC_API_KEY", + "ANTHROPIC_AUTH_TOKEN", + "ANTHROPIC_BASE_URL", + "ANTHROPIC_BEDROCK_BASE_URL", + "ANTHROPIC_FOUNDRY_BASE_URL", + "ANTHROPIC_VERTEX_BASE_URL", + "AWS_BEARER_TOKEN_BEDROCK", + "CLAUDE_CODE_USE_BEDROCK", + "CLAUDE_CODE_USE_FOUNDRY", + "CLAUDE_CODE_USE_VERTEX" + ] | sort) and all(.env[]; . == null)' \ + >/dev/null 2>&1 \ + || fail "authoritative injected --settings did not both null the credential keys and set feedbackDrafts off" + + ready="$dir/injected-ready" + FM_FAKE_STATE="$state" TEST_FAKE_SECRET="$SECRET" \ + "$AV_LAUNCH" --injected "$ready" "$fakebin/claude" "$injected_settings" \ + --settings '{"worker":"separated"}' --model sonnet \ + '--settings={"worker":"equals"}' launch-brief \ + 3>&1 4>&2 >/dev/null 2>&1 \ + || fail "direct injected launch with worker settings did not execute" + [ -e "$ready" ] || fail "direct injected launch did not signal readiness" + injected_argv=$(grep '^claude ' "$state/claude-argv.log" | tail -1) + settings_count=$(printf '%s\n' "$injected_argv" | grep -oE '<--settings(>|=)' | wc -l | tr -d ' ') + [ "$settings_count" = 1 ] \ + || fail "injected Claude exec carried $settings_count --settings flags after filtering both worker forms" + direct_settings=$(printf '%s\n' "$injected_argv" | sed -n 's/.*<--settings> <\([^>]*\)>.*/\1/p') + [ "$direct_settings" = "$injected_settings" ] \ + || fail "injected Claude exec replaced the authoritative settings with worker settings" + injected_worker_args=${injected_argv#*<--allowedTools> } + assert_contains "$injected_worker_args" '<--model> ' \ + "injected Claude exec dropped worker arguments adjacent to settings" + if printf '%s\n' "$injected_worker_args" | grep -qE '<--settings(>|=)|worker'; then + fail "worker settings survived into the injected Claude exec arguments" + fi + : > "$launchlog" record=$(make_ship "$dir" "$home" production-attestation) proj=${record%%$'\t'*}