From e9d8cc21cf6db5fd03e1bfd81b924f00f2a5526b Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Tue, 29 Sep 2026 21:45:11 +0800 Subject: [PATCH 01/21] Handle Lavish list-form feedback --- bin/fm-procevent-lavish.sh | 198 +++++++++++++----------- tests/fm-captain-hold-lifecycle.test.sh | 78 ++++++++++ 2 files changed, 187 insertions(+), 89 deletions(-) diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 81a38dac143..983876f935f 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -485,7 +485,7 @@ cmd_terminal() { # failed" is never proof that nothing was said. result_has_queued_content() { # awk ' - /^(prompts|feedback)\[[0-9]+\]\{[^}]*\}:[[:space:]]*$/ { + /^(prompts|feedback)\[[0-9]+\](\{[^}]*\})?:[[:space:]]*$/ { verdict = "present" exit } @@ -521,6 +521,93 @@ cmd_silent() { [ "$content_rc" -eq 1 ] } +# Parse either Lavish prompt representation into one JSON document. Uniform rows use +# a declared field list and CSV values; non-uniform rows use YAML-like list items and +# may contain nested attachment rows. The latter are metadata on the current item, +# never additional prompt items. +prompt_rows_json() { # + perl -MJSON::PP -e ' + use strict; use warnings; + my ($path) = @ARGV; + open my $fh, "<", $path or exit 1; + my ($declared, $header, $mode, $malformed) = (0, 0, "", 0); + my (@fields, @rows); + sub unquote { + my ($value) = @_; + $value =~ s/^\s+//; $value =~ s/\s+$//; + if ($value =~ /^"((?:[^"\\]|\\.)*)"$/s) { + $value = $1; + $value =~ s/\\(.)/$1 eq "n" ? "\n" : $1 eq "t" ? "\t" : $1 eq "r" ? "\r" : $1/ge; + } + return $value; + } + sub csv_values { + my ($row) = @_; + my @values; + while (length $row) { + if ($row =~ s/^"((?:[^"\\]|\\.)*)"//) { + push @values, unquote("\"$1\""); + } else { + $row =~ s/^([^,]*)//; + push @values, $1; + } + last unless $row =~ s/^,//; + } + return @values; + } + my ($current, $line); + while (defined($line = <$fh>)) { + if (!$header) { + if ($line =~ /^(?:prompts|feedback)\[(\d+)\]\{([^}]*)\}:\s*$/) { + ($declared, $header, $mode, @fields) = ($1, 1, "table", split /,/, $2); + next; + } + if ($line =~ /^(?:prompts|feedback)\[(\d+)\]:\s*$/) { + ($declared, $header, $mode) = ($1, 1, "list"); + next; + } + next; + } + if ($mode eq "table") { + last unless $line =~ /^\s/; + last if @rows >= $declared; + chomp $line; $line =~ s/^\s+//; + my @values = csv_values($line); + if (@values > @fields) { + my ($preserve) = grep { $fields[$_] eq "prompt" } 0 .. $#fields; + ($preserve) = grep { $fields[$_] eq "text" } 0 .. $#fields unless defined $preserve; + if (defined $preserve) { + my $count = @values - @fields; + my @parts = splice @values, $preserve, $count + 1; + splice @values, $preserve, 0, join(",", @parts); + } + } + if (@values != @fields) { $malformed++; next; } + my %row; $row{$fields[$_]} = $values[$_] for 0 .. $#fields; + push @rows, \%row; + next; + } + if ($line =~ /^ -\s+(.+)$/) { + push @rows, $current if defined $current; + $current = {}; + if ($1 =~ /^([A-Za-z_][A-Za-z0-9_]*):\s*(.*)$/) { + $current->{$1} = unquote($2); + } else { $malformed++; } + next; + } + if ($line =~ /^ ([A-Za-z_][A-Za-z0-9_]*):\s*(.*)$/) { + $current->{$1} = unquote($2) if defined $current; + next; + } + last if $line =~ /^\S/; + } + push @rows, $current if defined $current; + close $fh; + print encode_json({declared => $header ? 0 + $declared : 0, + rows => \@rows, malformed => 0 + $malformed, header => 0 + $header}); + ' "$1" +} + # Print `keyanswerlabel[mode]` for each non-reconcile structured choice the # captain submitted in a captured result; the optional mode column relays the # card's declared close mode (`done` or `release`) to the keyed-answer intake. The published response frames queued feedback as @@ -537,46 +624,20 @@ cmd_silent() { # `-decision-` identities pre-collapse decks still carry; the # security property is the slug SHAPE, which is unchanged. cmd_choice_rows() { - local selection=$1 file=${2-} + local selection=$1 file=${2-} parsed [ -n "$file" ] || usage [ -f "$file" ] && [ ! -L "$file" ] || die "result file does not exist: $file" - perl -MJSON::PP -MEncode=encode -e ' + parsed=$(prompt_rows_json "$file") || return 1 + printf '%s' "$parsed" | perl -MJSON::PP -MEncode=encode -e ' use strict; use warnings; - my ($selection, $path) = @ARGV; - open my $fh, "<", $path or exit 1; - my (@fields, $want, @rows); - while (my $line = <$fh>) { - if (!@fields) { - next unless $line =~ /^prompts\[(\d+)\]\{([^}]*)\}:\s*$/; - ($want, @fields) = ($1, split /,/, $2); - next; - } - last unless $line =~ /^\s/; - last if @rows >= $want; - chomp $line; - push @rows, $line; - } - close $fh; + my ($selection) = @ARGV; + my $doc = decode_json(do { local $/; }); + my @rows = @{$doc->{rows} || []}; my %seen; my @choices; - for my $row (@rows) { - $row =~ s/^\s+//; - my @vals; - while (length $row) { - if ($row =~ s/^"((?:[^"\\]|\\.)*)"//) { - my $v = $1; - $v =~ s/\\(.)/$1 eq "n" ? "\n" : $1 eq "t" ? "\t" : $1 eq "r" ? "\r" : $1/ge; - push @vals, $v; - } else { - $row =~ s/^([^,]*)//; - push @vals, $1; - } - last unless $row =~ s/^,//; - } - my %f; - $f{$fields[$_]} = $vals[$_] for 0 .. $#fields; - next unless defined $f{tag} && $f{tag} eq "choice"; - my $prompt = $f{prompt}; + for my $f (@rows) { + next unless defined $f->{tag} && $f->{tag} eq "choice"; + my $prompt = $f->{prompt}; next unless defined $prompt && $prompt =~ /Context data:\s*(\{.*\})/s; my $ctx = $1; my $data = eval { decode_json($ctx) }; @@ -616,7 +677,7 @@ cmd_choice_rows() { || ($data->{close} ne "done" && $data->{close} ne "release"); $mode = $data->{close}; } - my $label = defined $f{text} ? $f{text} : ""; + my $label = defined $f->{text} ? $f->{text} : ""; s/[\x00-\x1f\x7f]/ /g for ($answer, $note, $label); $label = substr($label, 0, 512); if (defined $seen{$key}) { $choices[$seen{$key}] = undef } @@ -643,7 +704,7 @@ cmd_choice_rows() { ? "$choice->{key}\t$answer\t$choice->{label}\t$choice->{mode}\n" : "$choice->{key}\t$answer\t$choice->{label}\n"; } - ' "$selection" "$file" + ' "$selection" } cmd_answers() { cmd_choice_rows answers "$@"; } @@ -658,61 +719,19 @@ cmd_reconciles() { cmd_choice_rows reconciles "$@"; } # comment matches the captured element text. Choice rows keep Context data # out of that field. A pure annotation has no prompt. cmd_read() { - local file=${1-} lifecycle session_ended + local file=${1-} lifecycle session_ended parsed [ -n "$file" ] || usage [ -f "$file" ] && [ ! -L "$file" ] || die "result file does not exist: $file" lifecycle=$(cmd_classify "$file") session_ended=$(session_field "$file" session_ended) - perl -e ' + parsed=$(prompt_rows_json "$file") || return 1 + printf '%s' "$parsed" | perl -MJSON::PP -e ' use strict; use warnings; - my ($path, $lifecycle, $session_ended) = @ARGV; - open my $fh, "<", $path or exit 1; - my (@fields, $want, @rows); - while (my $line = <$fh>) { - if (!@fields) { - next unless $line =~ /^(?:prompts|feedback)\[(\d+)\]\{([^}]*)\}:\s*$/; - ($want, @fields) = ($1, split /,/, $2); - next; - } - last unless $line =~ /^\s/; - last if defined($want) && @rows >= $want; - chomp $line; - push @rows, $line; - } - close $fh; - $want = 0 unless defined $want; - my @parsed; - my $malformed = 0; - for my $row (@rows) { - $row =~ s/^\s+//; - my @vals; - while (length $row) { - if ($row =~ s/^"((?:[^"\\]|\\.)*)"//) { - push @vals, $1; - } else { - $row =~ s/^([^,]*)//; - push @vals, $1; - } - last unless $row =~ s/^,//; - } - if (@vals > @fields) { - my ($preserve) = grep { $fields[$_] eq "prompt" } 0 .. $#fields; - ($preserve) = grep { $fields[$_] eq "text" } 0 .. $#fields unless defined $preserve; - if (defined $preserve) { - my $count = @vals - @fields + 1; - my @parts = splice @vals, $preserve, $count; - splice @vals, $preserve, 0, join(",", @parts); - } - } - if (@vals != @fields) { - $malformed++; - next; - } - s/\\(.)/$1 eq "n" ? "\n" : $1 eq "t" ? "\t" : $1 eq "r" ? "\r" : $1/ge for @vals; - my %f; - $f{$fields[$_]} = $vals[$_] for 0 .. $#fields; - push @parsed, \%f; - } + my ($lifecycle, $session_ended) = @ARGV; + my $doc = decode_json(do { local $/; }); + my $want = $doc->{declared} || 0; + my @parsed = @{$doc->{rows} || []}; + my $malformed = $doc->{malformed} || 0; my $presented = scalar @parsed; my $complete = ($presented == $want && !$malformed) ? "yes" : "no"; my @messages; @@ -787,7 +806,8 @@ cmd_read() { print "ANNOTATIONS: (none)\n"; } print "END LAVISH RESULT ($presented of $want)\n"; - ' "$file" "$lifecycle" "$session_ended" + exit($complete eq "yes" ? 0 : 1); + ' "$lifecycle" "$session_ended" } case "${1-}" in diff --git a/tests/fm-captain-hold-lifecycle.test.sh b/tests/fm-captain-hold-lifecycle.test.sh index 0267d97efd1..40967787fdd 100755 --- a/tests/fm-captain-hold-lifecycle.test.sh +++ b/tests/fm-captain-hold-lifecycle.test.sh @@ -4016,6 +4016,83 @@ PM pass "both body-decoding paths work without the allow_nonref default" } +test_lavish_list_form_feedback_is_complete() { + local home result mismatch out rc silent_rc + home=$(make_home lavish-list-form) + result="$home/list-form.result" + cat > "$result" <<'EOF' + session: + status: feedback + session_ended: true + prompts[5]: + - uid: "1" + prompt: "A free comment" + selector: "#comment" + tag: p + text: "Element text" + - uid: "2" + prompt: "Choice\n\nContext data:\n{\n \"schema\": \"fm-bearings-answer.v1\",\n \"question\": \"list-choice\",\n \"selection\": \"yes\",\n \"note\": \"choice note\"\n}" + selector: "#choice" + tag: choice + text: "Choose yes" + - uid: "3" + prompt: "Attached comment" + selector: "#attached" + tag: p + text: "Attached element" + attachments[1]{id,type}: + attachment-id,image + - uid: "4" + prompt: "Session says stop" + selector: "" + tag: message + text: "Session message" + - uid: "5" + prompt: "Another comment" + selector: "#another" + tag: p + text: "Another element" + next_step: done +EOF + # The leading spaces above are intentional YAML-like fixture indentation. + # Normalize only the block indentation so the adapter sees the published shape. + perl -pi -e 's/^ //' "$result" + + set +e + out=$(run_lavish "$home" read "$result" 2>&1) + rc=$? + set -e + [ "$rc" -eq 0 ] || fail "complete list-form feedback was rejected: $out" + assert_contains "$out" "declared_items: 5" "list-form declared count was lost" + assert_contains "$out" "presented_items: 5" "list-form items were dropped" + assert_contains "$out" "complete: yes" "complete list-form feedback was marked incomplete" + assert_contains "$out" "annotation_count: 4" "list-form annotation count was wrong" + assert_contains "$out" "| A free comment" "freeform annotation comment was lost" + assert_contains "$out" "| Attached comment" "annotation with an attachment was lost" + assert_contains "$out" "SESSION-ENDING MESSAGE" "session-ending message was not presented" + assert_contains "$out" "| Session says stop" "session-ending message body was lost" + + out=$(run_lavish "$home" answers "$result") || fail "list-form choice answer could not be read" + assert_contains "$out" $'list-choice\tyes - choice note' "list-form Context data answer was lost" + + silent_rc=0 + run_lavish "$home" silent "$result" >/dev/null 2>&1 || silent_rc=$? + [ "$silent_rc" -ne 0 ] || fail "list-form feedback was incorrectly treated as silent" + + mismatch="$home/list-form-mismatch.result" + perl -pe 's/prompts\[5\]:/prompts[6]:/' "$result" > "$mismatch" + set +e + out=$(run_lavish "$home" read "$mismatch" 2>&1) + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "declared list-form count mismatch reported success: $out" + assert_contains "$out" "complete: no" "declared list-form count mismatch was marked complete" + silent_rc=0 + run_lavish "$home" silent "$mismatch" >/dev/null 2>&1 || silent_rc=$? + [ "$silent_rc" -ne 0 ] || fail "nonzero declared list-form feedback was treated as silent" + pass "Lavish list-form feedback preserves annotations, choices, attachments, and incomplete captures" +} + # Cleanup rewrites a captain-held row's body to append the finished work's # deliverable, so every byte of that body has to survive the decode. The # assertions below are on bytes, not characters: a decoder that prints a @@ -4081,6 +4158,7 @@ test_retained_body_keeps_its_utf8_bytes() { test_uninventoried_report_decision_refuses_completion test_hold_decodes_a_bare_scalar_body_without_the_nonref_default +test_lavish_list_form_feedback_is_complete test_retained_body_keeps_its_utf8_bytes test_completion_gate_attests_and_transfers test_answer_records_and_closes From 9147025694656da71ebcd4a5bbd4dfbaafd25a22 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Tue, 29 Sep 2026 21:53:42 +0800 Subject: [PATCH 02/21] no-mistakes(review): Parse table rows beyond declared counts --- bin/fm-procevent-lavish.sh | 1 - tests/fm-procevent.test.sh | 28 ++++++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 983876f935f..53e4a20735e 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -570,7 +570,6 @@ prompt_rows_json() { # } if ($mode eq "table") { last unless $line =~ /^\s/; - last if @rows >= $declared; chomp $line; $line =~ s/^\s+//; my @values = csv_values($line); if (@values > @fields) { diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index eb511b588fb..94672a0fca8 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -3098,6 +3098,34 @@ ann_line=$(printf '%s\n' "$out" | grep -n '^ANNOTATIONS$' | head -1 | cut -d: -f || fail "the item count did not appear before the annotations" pass "read presents every annotation and a distinct session-ending message" +cat > "$READ" <<'EOF' +session: + file: /review.html + status: feedback +prompts[1]{uid,prompt,selector,tag,text}: + "el-a","first comment","section#first",note,"First item" + "el-b","Context data: {\"schema\":\"fm-bearings-answer.v1\",\"question\":\"overflow-answer\",\"selection\":\"yes\",\"note\":\"\"}","section#answer",choice,"Yes" + "el-c","Context data: {\"schema\":\"fm-bearings-answer.v1\",\"question\":\"overflow-reconcile\",\"selection\":\"reconcile\",\"note\":\"check extra\"}","section#reconcile",choice,"Reconcile" +EOF +read_status=0 +out=$(read_out 2>&1) || read_status=$? +[ "$read_status" -ne 0 ] || fail "read certified more table rows than declared as complete" +assert_contains "$out" "declared_items: 1" "an overfull table lost its declared count" +assert_contains "$out" "presented_items: 3" "read discarded table rows beyond the declared count" +assert_contains "$out" "complete: no" "an overfull table was certified as complete" +assert_contains "$out" "| First item" "read dropped the first row from an overfull table" +assert_contains "$out" "| Yes" "read dropped an answer row beyond the declared count" +assert_contains "$out" "| Reconcile" "read dropped a reconcile row beyond the declared count" +out=$("$ROOT/bin/fm-procevent-lavish.sh" answers "$READ") \ + || fail "answers failed on an overfull table" +[ "$out" = "$(printf 'overflow-answer\tyes\tYes')" ] \ + || fail "answers lost or invented rows in an overfull table: $out" +out=$("$ROOT/bin/fm-procevent-lavish.sh" reconciles "$READ") \ + || fail "reconciles failed on an overfull table" +[ "$out" = "$(printf 'overflow-reconcile\tcheck extra')" ] \ + || fail "reconciles lost or invented rows in an overfull table: $out" +pass "table parsing reports and preserves rows beyond the declared count" + cat > "$READ" <<'EOF' session: file: /review.html From c1e01b4443363d11d91f485bf505b444041b6cf3 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Tue, 29 Sep 2026 22:25:43 +0800 Subject: [PATCH 03/21] no-mistakes(document): Document Lavish list-form feedback parsing --- .agents/skills/process-event-sources/SKILL.md | 1 + bin/fm-procevent-lavish.sh | 42 ++++++++++--------- docs/verification/process-event-sources.md | 1 + 3 files changed, 25 insertions(+), 19 deletions(-) diff --git a/.agents/skills/process-event-sources/SKILL.md b/.agents/skills/process-event-sources/SKILL.md index 8b765f01c8a..8a72732e243 100644 --- a/.agents/skills/process-event-sources/SKILL.md +++ b/.agents/skills/process-event-sources/SKILL.md @@ -114,6 +114,7 @@ Two rules the commands cannot enforce for you: : Ask the adapter what the result means rather than parsing it yourself. `bin/fm-procevent.sh classify ` routes through the immutable built-in or extension identity captured with that result; for Lavish, its existing direct command returns `feedback`, `ended`, `waiting`, `disconnected`, `missing`, or `unknown`. Consume a Lavish capture with `bin/fm-procevent-lavish.sh read ` rather than grepping the raw file: that command reports declared and presented item counts plus a completeness verdict, enumerates every captured queued item while retaining supplied element identity, and surfaces a `tag=message` freeform message as its own field, labeling it as session-ending only when the session ended. + A count mismatch or malformed item makes `read` report an incomplete result and exit nonzero, so leave that capture unacknowledged. `answers` remains the keyed-choice extractor and never treats freeform prose as a decision key. A `feedback` result can still be the last one a review ever produces, so never assume another wake is coming just because the state is not `ended`. The crew-hosted recovery ordering and arm-and-acknowledge rule are owned by the [crew-hosted Lavish board contract](../../../docs/configuration.md#crew-hosted-lavish-review-boards); `bin/fm-brief.sh` emits its instruction at the point of use. diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 53e4a20735e..1ed9d85791f 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -21,9 +21,12 @@ # what Lavish delivered. The freeform message (tag=message) is its # own labeled field, printed first and distinct from per-element # annotations; it is labeled SESSION-ENDING MESSAGE only when the -# session ended. Declared and presented item counts, -# plus a completeness verdict, follow before all annotations so a -# partial read is obvious. Each annotation retains its element uid, +# session ended. It accepts both field-declared CSV tables and +# YAML-like item lists, including list items with nested attachment +# metadata. Declared and presented item counts, plus a completeness +# verdict, follow before all annotations so a partial read is +# obvious. A count mismatch or malformed row prints `complete: no` +# and exits nonzero. Each annotation retains its element uid, # selector, tag, and text. A non-choice freeform comment (`prompt`) # is printed as its own field even when a selector is also present # and even when that comment matches the element text, so typed @@ -473,12 +476,13 @@ cmd_terminal() { } # Whether a completed result carries any queued content block at all. The -# published response frames content as a top-level `prompts[N]{...}:` or -# `feedback[N]{...}:` header whose rows are INDENTED, so this anchors on column -# zero: an indented payload line is captain-supplied text and must never be able -# to forge - or, here, to hide behind - a content header. Any recognized block -# is content regardless of its declared count, while a malformed top-level -# prompts or feedback header makes the result indeterminate. +# published response frames content as a top-level `prompts[N]:` or +# `feedback[N]:` list, or as the field-declared table variant +# `prompts[N]{...}:` or `feedback[N]{...}:`. This check anchors on column zero: +# an indented payload line is captain-supplied text and must never be able to +# forge - or, here, to hide behind - a content header. Any recognized block is +# content regardless of its declared count, while a malformed top-level prompts +# or feedback header makes the result indeterminate. # # 0 = content present, 1 = provably no content, anything else = the check did # not complete. The caller must distinguish those three, because "the check @@ -521,10 +525,10 @@ cmd_silent() { [ "$content_rc" -eq 1 ] } -# Parse either Lavish prompt representation into one JSON document. Uniform rows use -# a declared field list and CSV values; non-uniform rows use YAML-like list items and -# may contain nested attachment rows. The latter are metadata on the current item, -# never additional prompt items. +# Parse either Lavish prompt representation into one JSON document. +# Uniform rows use a declared field list and CSV values. +# Non-uniform rows use YAML-like list items and may contain nested attachment +# rows, which are metadata on the current item rather than more prompt items. prompt_rows_json() { # perl -MJSON::PP -e ' use strict; use warnings; @@ -537,8 +541,8 @@ prompt_rows_json() { # $value =~ s/^\s+//; $value =~ s/\s+$//; if ($value =~ /^"((?:[^"\\]|\\.)*)"$/s) { $value = $1; - $value =~ s/\\(.)/$1 eq "n" ? "\n" : $1 eq "t" ? "\t" : $1 eq "r" ? "\r" : $1/ge; } + $value =~ s/\\(.)/$1 eq "n" ? "\n" : $1 eq "t" ? "\t" : $1 eq "r" ? "\r" : $1/ge; return $value; } sub csv_values { @@ -549,7 +553,7 @@ prompt_rows_json() { # push @values, unquote("\"$1\""); } else { $row =~ s/^([^,]*)//; - push @values, $1; + push @values, unquote($1); } last unless $row =~ s/^,//; } @@ -609,10 +613,10 @@ prompt_rows_json() { # # Print `keyanswerlabel[mode]` for each non-reconcile structured choice the # captain submitted in a captured result; the optional mode column relays the -# card's declared close mode (`done` or `release`) to the keyed-answer intake. The published response frames queued feedback as -# a `prompts[N]{field,...}:` header followed by exactly N indented CSV rows whose -# quoted fields carry JSON-style escapes, so this reads the declared field ORDER -# rather than assuming a fixed column, and takes only rows whose `tag` field is +# card's declared close mode (`done` or `release`) to the keyed-answer intake. +# Queued feedback can use field-declared CSV rows or YAML-like list items, so the +# shared parser normalizes both forms and preserves every parsed item even when +# the declared count differs. This command takes only rows whose `tag` field is # `choice`. A freeform `message` row is captain prose and is deliberately never a # source of decision keys. A row that does not carry both a slug-shaped `question` # and the versioned `selection` and `note` fields inside its `Context data:` block diff --git a/docs/verification/process-event-sources.md b/docs/verification/process-event-sources.md index 8abe4a71a06..1dcfa5fa3ae 100644 --- a/docs/verification/process-event-sources.md +++ b/docs/verification/process-event-sources.md @@ -104,6 +104,7 @@ Exercised by `tests/fm-procevent.test.sh` against a fake blocking source whose c | adapter-owned silence verdict | an ordinary firstmate-owned Lavish source driven against a stand-in poll that returns an empty ended session captures its result, records it durably handled, appends no wake, and stays silent through a later `reconcile` that would otherwise republish it, while still retiring its ended source; the same real path with a `Send & End` response carrying the captain's choice still publishes its `check` wake and is left unacknowledged for the handler | | worker-owned Lavish rounds | one three-round fixture arms a board for an identity-matched task endpoint, delivers nonterminal and terminal captures directly to that task's steering inbox without a firstmate `check` wake, acknowledges each nonterminal round through a successful re-arm, redelivers an inbox note filed before acknowledgement, refuses a second armer and every early retirement, and concludes the terminal round through `handled` without another poll; focused fixtures also pin failed re-arm rollback, generation-specific reply staging, one reply post across transient poll retries, unreachable-owner refusal, interrupted conclusion recovery, and repeat acknowledgement isolation | | Lavish handled-status classification | an executable fixture table pins exact `feedback`, `ended`, `waiting`, and `browser_disconnected` mappings, including `browser_disconnected` to `disconnected`; the same suite proves that status is nonterminal and receives a zero-answer silence verdict | +| Lavish result decoding | `tests/fm-procevent.test.sh` and `tests/fm-captain-hold-lifecycle.test.sh` use synthetic field-declared tables and YAML-like item lists to prove that `read`, `answers`, `reconciles`, and `silent` see every item; list coverage includes annotations, versioned choice context, nested attachment metadata, and a session-ending message, while both representations prove that declared-count mismatches preserve parsed rows, report incomplete, and make `read` exit nonzero | | session-derived Lavish routing | the three-round worker fixture starts its first listener under conflicting ambient host/port values and configuration, then recovers later listeners while that conflicting configuration remains, and proves every reply/poll uses the board's saved session endpoint; direct polls cover Unicode artifact paths, hostnames, IPv6, session endpoint changes, quiet retries, and refusal before reply consumption when session evidence is absent or invalid; spawn coverage still proves the configured opening address enters the worker launch | | silence fails closed | the adapter's published `silent` command suppresses only an `ended` session with no queued content block or a `browser_disconnected` response, and announces a real answer, freeform prose, any recognized content block regardless of its declared count, a malformed top-level content header, a `waiting` or `missing` session, a server error, an unreadable result, and indented payload text imitating an empty content block; the `remote-reply` and `when` adapters, which implement no `silent` command, announce every result | | terminal retirement preserves the result | the retired source's captured output, its announced event, its handled acknowledgement, and later explicit `retire` all still behave normally | From e66a604a0eae13f2d2fee0d8292ca3e88a11bd0e Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Tue, 29 Sep 2026 22:56:30 +0800 Subject: [PATCH 04/21] no-mistakes(ci): Updated the malformed-capture regression to require read's nonzero incomplete verdict while retaining output assertions. `tests/fm-procevent.test.sh`, `tests/fm-bearings-board-render.test.sh`, `bin/fm-lint.sh`, and `git diff --check` pass. Serial 8 was a transient runner failure and passed six local runs --- tests/fm-procevent.test.sh | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index 94672a0fca8..38d6050f114 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -3136,7 +3136,9 @@ prompts[2]{uid,prompt,selector,tag,text}: "el-a","","section#call",note,"Complete annotation" "el-b","","section#other",note EOF -out=$(read_out) || fail "read failed on a capture containing a malformed item" +read_status=0 +out=$(read_out 2>&1) || read_status=$? +[ "$read_status" -ne 0 ] || fail "read certified a malformed capture as complete" assert_contains "$out" "declared_items: 2" "a malformed capture lost its declared count" assert_contains "$out" "presented_items: 1" \ "a row missing declared fields was certified as presented" From 2a37594ceb283559648cbf03679787fb68bf4f52 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Tue, 29 Sep 2026 23:43:01 +0800 Subject: [PATCH 05/21] Preserve Lavish UTF-8 feedback --- bin/fm-procevent-lavish.sh | 11 ++++++---- tests/fm-procevent.test.sh | 45 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 4 deletions(-) diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 1ed9d85791f..a2cd7d5e11b 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -533,7 +533,7 @@ prompt_rows_json() { # perl -MJSON::PP -e ' use strict; use warnings; my ($path) = @ARGV; - open my $fh, "<", $path or exit 1; + open my $fh, "<:encoding(UTF-8)", $path or exit 1; my ($declared, $header, $mode, $malformed) = (0, 0, "", 0); my (@fields, @rows); sub unquote { @@ -633,6 +633,7 @@ cmd_choice_rows() { parsed=$(prompt_rows_json "$file") || return 1 printf '%s' "$parsed" | perl -MJSON::PP -MEncode=encode -e ' use strict; use warnings; + binmode STDOUT, ":raw"; my ($selection) = @ARGV; my $doc = decode_json(do { local $/; }); my @rows = @{$doc->{rows} || []}; @@ -643,7 +644,7 @@ cmd_choice_rows() { my $prompt = $f->{prompt}; next unless defined $prompt && $prompt =~ /Context data:\s*(\{.*\})/s; my $ctx = $1; - my $data = eval { decode_json($ctx) }; + my $data = eval { decode_json(encode("UTF-8", $ctx)) }; next unless ref($data) eq "HASH"; my ($key, $selected, $note, $answer, $legacy); if (defined($data->{schema}) && !ref($data->{schema}) @@ -703,9 +704,10 @@ cmd_choice_rows() { } next if $choice->{selection} eq "reconcile"; my $answer = encode("UTF-8", $choice->{answer}); + my $label = encode("UTF-8", $choice->{label}); print length $choice->{mode} - ? "$choice->{key}\t$answer\t$choice->{label}\t$choice->{mode}\n" - : "$choice->{key}\t$answer\t$choice->{label}\n"; + ? "$choice->{key}\t$answer\t$label\t$choice->{mode}\n" + : "$choice->{key}\t$answer\t$label\n"; } ' "$selection" } @@ -730,6 +732,7 @@ cmd_read() { parsed=$(prompt_rows_json "$file") || return 1 printf '%s' "$parsed" | perl -MJSON::PP -e ' use strict; use warnings; + binmode STDOUT, ":encoding(UTF-8)"; my ($lifecycle, $session_ended) = @ARGV; my $doc = decode_json(do { local $/; }); my $want = $doc->{declared} || 0; diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index 38d6050f114..18446e3a17c 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -3126,6 +3126,51 @@ out=$("$ROOT/bin/fm-procevent-lavish.sh" reconciles "$READ") \ || fail "reconciles lost or invented rows in an overfull table: $out" pass "table parsing reports and preserves rows beyond the declared count" +for shape in table list; do + if [ "$shape" = table ]; then + cat > "$READ" <<'EOF' +session: + status: feedback +prompts[3]{uid,prompt,selector,tag,text}: + "el-a","Comment café","section#comment",note,"Element café" + "el-b","Context data: {\"schema\":\"fm-bearings-answer.v1\",\"question\":\"unicode-answer\",\"selection\":\"yes\",\"note\":\"東京\"}","section#answer",choice,"Answer 東京" + "el-c","Context data: {\"schema\":\"fm-bearings-answer.v1\",\"question\":\"unicode-reconcile\",\"selection\":\"reconcile\",\"note\":\"réexaminer café\"}","section#reconcile",choice,"Reconcile" +EOF + else + cat > "$READ" <<'EOF' +session: + status: feedback +prompts[3]: + - uid: "el-a" + prompt: "Comment café" + selector: "section#comment" + tag: note + text: "Element café" + - uid: "el-b" + prompt: "Context data: {\"schema\":\"fm-bearings-answer.v1\",\"question\":\"unicode-answer\",\"selection\":\"yes\",\"note\":\"東京\"}" + selector: "section#answer" + tag: choice + text: "Answer 東京" + - uid: "el-c" + prompt: "Context data: {\"schema\":\"fm-bearings-answer.v1\",\"question\":\"unicode-reconcile\",\"selection\":\"reconcile\",\"note\":\"réexaminer café\"}" + selector: "section#reconcile" + tag: choice + text: "Reconcile" +EOF + fi + out=$(read_out) || fail "${shape}-form Unicode read failed" + assert_contains "$out" "Comment café" "${shape}-form Unicode comment was lost" + out=$("$ROOT/bin/fm-procevent-lavish.sh" answers "$READ") \ + || fail "${shape}-form Unicode answers failed" + [ "$out" = "$(printf 'unicode-answer\tyes - 東京\tAnswer 東京')" ] \ + || fail "${shape}-form Unicode answer was corrupted: $out" + out=$("$ROOT/bin/fm-procevent-lavish.sh" reconciles "$READ") \ + || fail "${shape}-form Unicode reconciles failed" + [ "$out" = "$(printf 'unicode-reconcile\tréexaminer café')" ] \ + || fail "${shape}-form Unicode reconcile note was corrupted: $out" +done +pass "Lavish table and list forms preserve Unicode comments, answers, and reconcile notes" + cat > "$READ" <<'EOF' session: file: /review.html From 8a87193bb078bdb7481095bbc7764c2e0703dd0e Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Wed, 30 Sep 2026 00:26:46 +0800 Subject: [PATCH 06/21] no-mistakes(document): Document Lavish result decoding guarantees --- bin/fm-procevent-lavish.sh | 2 ++ docs/verification/process-event-sources.md | 2 +- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index a2cd7d5e11b..fe4937e2ba9 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -529,6 +529,8 @@ cmd_silent() { # Uniform rows use a declared field list and CSV values. # Non-uniform rows use YAML-like list items and may contain nested attachment # rows, which are metadata on the current item rather than more prompt items. +# Decode the capture as UTF-8 once here; output consumers encode text as UTF-8 +# once so comments, answer labels, and reconcile notes keep their bytes. prompt_rows_json() { # perl -MJSON::PP -e ' use strict; use warnings; diff --git a/docs/verification/process-event-sources.md b/docs/verification/process-event-sources.md index 1dcfa5fa3ae..50d7c60255e 100644 --- a/docs/verification/process-event-sources.md +++ b/docs/verification/process-event-sources.md @@ -104,7 +104,7 @@ Exercised by `tests/fm-procevent.test.sh` against a fake blocking source whose c | adapter-owned silence verdict | an ordinary firstmate-owned Lavish source driven against a stand-in poll that returns an empty ended session captures its result, records it durably handled, appends no wake, and stays silent through a later `reconcile` that would otherwise republish it, while still retiring its ended source; the same real path with a `Send & End` response carrying the captain's choice still publishes its `check` wake and is left unacknowledged for the handler | | worker-owned Lavish rounds | one three-round fixture arms a board for an identity-matched task endpoint, delivers nonterminal and terminal captures directly to that task's steering inbox without a firstmate `check` wake, acknowledges each nonterminal round through a successful re-arm, redelivers an inbox note filed before acknowledgement, refuses a second armer and every early retirement, and concludes the terminal round through `handled` without another poll; focused fixtures also pin failed re-arm rollback, generation-specific reply staging, one reply post across transient poll retries, unreachable-owner refusal, interrupted conclusion recovery, and repeat acknowledgement isolation | | Lavish handled-status classification | an executable fixture table pins exact `feedback`, `ended`, `waiting`, and `browser_disconnected` mappings, including `browser_disconnected` to `disconnected`; the same suite proves that status is nonterminal and receives a zero-answer silence verdict | -| Lavish result decoding | `tests/fm-procevent.test.sh` and `tests/fm-captain-hold-lifecycle.test.sh` use synthetic field-declared tables and YAML-like item lists to prove that `read`, `answers`, `reconciles`, and `silent` see every item; list coverage includes annotations, versioned choice context, nested attachment metadata, and a session-ending message, while both representations prove that declared-count mismatches preserve parsed rows, report incomplete, and make `read` exit nonzero | +| Lavish result decoding | `tests/fm-procevent.test.sh` and `tests/fm-captain-hold-lifecycle.test.sh` use synthetic field-declared tables and YAML-like item lists to prove that `read` preserves every parsed row, `answers` and `reconciles` extract their respective choice rows from either form, and `silent` recognizes either representation as content; list coverage includes annotations, versioned choice context, nested attachment metadata, and a session-ending message; both representations preserve non-ASCII comments, answers, and reconcile notes; declared-count mismatches preserve parsed rows, report incomplete, and make `read` exit nonzero; a malformed table row also reports incomplete without hiding its valid neighbor | | session-derived Lavish routing | the three-round worker fixture starts its first listener under conflicting ambient host/port values and configuration, then recovers later listeners while that conflicting configuration remains, and proves every reply/poll uses the board's saved session endpoint; direct polls cover Unicode artifact paths, hostnames, IPv6, session endpoint changes, quiet retries, and refusal before reply consumption when session evidence is absent or invalid; spawn coverage still proves the configured opening address enters the worker launch | | silence fails closed | the adapter's published `silent` command suppresses only an `ended` session with no queued content block or a `browser_disconnected` response, and announces a real answer, freeform prose, any recognized content block regardless of its declared count, a malformed top-level content header, a `waiting` or `missing` session, a server error, an unreadable result, and indented payload text imitating an empty content block; the `remote-reply` and `when` adapters, which implement no `silent` command, announce every result | | terminal retirement preserves the result | the retired source's captured output, its announced event, its handled acknowledgement, and later explicit `retire` all still behave normally | From c25c34e74d22a19f7db2b0559c4c436596977f5b Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:54:08 +0800 Subject: [PATCH 07/21] Pass supported Codex max effort to workers --- .../references/harness/codex.md | 2 +- bin/fm-spawn.sh | 25 ++++++++++--- tests/fm-secondmate-harness.test.sh | 35 +++++++++++++++++++ 3 files changed, 56 insertions(+), 6 deletions(-) diff --git a/.agents/skills/harness-adapters/references/harness/codex.md b/.agents/skills/harness-adapters/references/harness/codex.md index d68486f12e2..cba029962e7 100644 --- a/.agents/skills/harness-adapters/references/harness/codex.md +++ b/.agents/skills/harness-adapters/references/harness/codex.md @@ -12,7 +12,7 @@ Verified on 2026-06-11 with codex-cli 0.139.0 unless a fact gives a newer versio | Skill invocation | `$`, for example `$no-mistakes`; `/` is Claude-only and Codex rejects it as "Unrecognized command". | | Resume | `codex resume `, using the id printed on quit. | | Model flag | `--model `. | -| Effort flag | `-c 'model_reasoning_effort=""'`, verified on codex-cli 0.142.1 whose installed schema contains `model_reasoning_effort`, active config uses it, and bundled catalog advertised only the first four values while omitting `max`; current codex-cli 0.153.4 catalog data at `${CODEX_HOME:-~/.codex}/models_cache.json` advertises `max` for `gpt-5.6-luna`, which Firstmate passes for that model. | +| Effort flag | `-c 'model_reasoning_effort=""'`; for `max`, Firstmate passes the setting only when the selected model's entry in `${CODEX_HOME:-~/.codex}/models_cache.json` advertises `max`, otherwise it warns and omits it. | | Model discovery | Open the current interactive session's `/model` picker. | | Marker | None; identity comes from ancestry, and `../../../bin/fm-harness.sh` is what keeps a retained foreign `CLAUDECODE` from renaming it. Verified on 2026-09-01 with codex-cli 0.152.0: the pane process is the `node` npm shim and the native `codex` binary runs as its foreground child, so a tool subprocess reaches the native name directly while the shim itself is identified from its script path. | diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 764109f749b..371986c9fd3 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -2338,6 +2338,18 @@ model_flag_for_harness() { esac } +codex_catalog_supports_effort() { + local model=$1 effort=$2 catalog="${CODEX_HOME:-$HOME/.codex}/models_cache.json" + [ -n "$model" ] && [ "$model" != default ] || return 1 + command -v jq >/dev/null 2>&1 || return 1 + jq -e --arg model "$model" --arg effort "$effort" ' + any(.models[]?; + ((.slug? // .id? // .model?) == $model) + and any(.supported_reasoning_levels[]?; .effort? == $effort) + ) + ' "$catalog" >/dev/null 2>&1 +} + effort_flag_for_harness() { local harness=$1 effort=$2 model=${3:-} [ -n "$effort" ] && [ "$effort" != default ] || return 0 @@ -2348,14 +2360,17 @@ effort_flag_for_harness() { esac ;; codex) - # The installed codex config schema uses model_reasoning_effort. The - # installed model catalog supports max for gpt-5.6-luna; keep that level - # scoped to the model whose catalog entry advertises it. + # Codex exposes model_reasoning_effort only for levels advertised by the + # selected model's installed catalog entry. case "$effort" in low | medium | high | xhigh) printf -- '-c %s ' "$(shell_quote "model_reasoning_effort=\"$effort\"")" ;; max) - [ "$model" = gpt-5.6-luna ] || return 0 - printf -- '-c %s ' "$(shell_quote 'model_reasoning_effort="max"')" + if codex_catalog_supports_effort "$model" "$effort"; then + printf -- '-c %s ' "$(shell_quote 'model_reasoning_effort="max"')" + else + printf 'warning: dropped codex effort %s for model %s; catalog does not advertise it\n' \ + "$effort" "${model:-default}" >&2 + fi ;; esac ;; diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index 6b98ebd4426..d2c7c30d178 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -888,6 +888,40 @@ test_spawn_explicit_harness_does_not_inherit_secondmate_harness_tokens() { pass "C7 spawn: an explicit --harness starts with clean model/effort defaults" } +test_codex_catalog_max_effort() { + local w sm launchlog codex_home launch w2 sm2 launchlog2 err + w="$TMP_ROOT/spawn-codex-catalog-max" + w2="$TMP_ROOT/spawn-codex-catalog-missing" + sm="$w/sm" + launchlog="$w/launch.log" + codex_home="$w/codex" + mkdir -p "$w/home/config" "$codex_home" + printf 'codex\n' > "$w/home/config/secondmate-harness" + printf '%s\n' '{"models":[{"slug":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ + > "$codex_home/models_cache.json" + make_seeded_home "$sm" sm + + CODEX_HOME="$codex_home" spawn_secondmate_capture \ + "$w" sm "$sm" "$launchlog" --harness codex --model gpt-6-astra --effort max \ + >/dev/null 2>&1 + + launch=$(cat "$launchlog") + assert_contains "$launch" "-c 'model_reasoning_effort=\"max\"'" \ + "codex catalog max effort was not passed to the launch" + + sm2="$w2/sm" + launchlog2="$w2/launch.log" + err="$w2/stderr" + mkdir -p "$w2/home/config" + make_seeded_home "$sm2" sm + CODEX_HOME="$w2/missing" spawn_secondmate_capture \ + "$w2" sm "$sm2" "$launchlog2" --harness codex --model gpt-6-astra --effort max \ + >/dev/null 2>"$err" + assert_contains "$(cat "$err")" "dropped codex effort max" \ + "missing Codex catalog did not warn about the dropped effort" + pass "Codex max effort follows the selected model catalog and warns when absent" +} + test_spawn_explicit_harness_uses_explicit_profile_axes() { local w sm meta launchlog launch w="$TMP_ROOT/spawn-explicit-harness-explicit-axes" @@ -2655,6 +2689,7 @@ test_spawn_secondmate_harness_model_and_effort_tokens test_spawn_explicit_model_overrides_secondmate_harness_token test_spawn_explicit_effort_overrides_secondmate_harness_token test_spawn_explicit_harness_does_not_inherit_secondmate_harness_tokens +test_codex_catalog_max_effort test_spawn_explicit_harness_uses_explicit_profile_axes test_spawned_secondmate_uses_its_harness_supervision_model test_spawn_fallback_chain_and_crew_scout_unaffected From 4ea5d84208a726ce27bb97e63a9535912a19200c Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:26:02 +0800 Subject: [PATCH 08/21] no-mistakes(review): Reject malformed Lavish lists and isolate Codex catalogs --- bin/fm-procevent-lavish.sh | 32 +++++++++++++++++++++++-- tests/fm-procevent.test.sh | 21 ++++++++++++++++ tests/fm-spawn-dispatch-profile.test.sh | 24 ++++++++++++++----- 3 files changed, 69 insertions(+), 8 deletions(-) diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index fe4937e2ba9..2a846213367 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -561,7 +561,7 @@ prompt_rows_json() { # } return @values; } - my ($current, $line); + my ($current, $line, $attachment_rows, $attachment_fields); while (defined($line = <$fh>)) { if (!$header) { if ($line =~ /^(?:prompts|feedback)\[(\d+)\]\{([^}]*)\}:\s*$/) { @@ -592,6 +592,11 @@ prompt_rows_json() { # push @rows, \%row; next; } + if (defined($attachment_rows) && $line !~ /^ /) { + $malformed++ if $attachment_rows; + undef $attachment_rows; + undef $attachment_fields; + } if ($line =~ /^ -\s+(.+)$/) { push @rows, $current if defined $current; $current = {}; @@ -601,11 +606,34 @@ prompt_rows_json() { # next; } if ($line =~ /^ ([A-Za-z_][A-Za-z0-9_]*):\s*(.*)$/) { - $current->{$1} = unquote($2) if defined $current; + if (defined $current) { + $current->{$1} = unquote($2); + } else { $malformed++; } + next; + } + if ($line =~ /^ attachments\[(\d+)\]\{([A-Za-z_][A-Za-z0-9_]*(?:,[A-Za-z_][A-Za-z0-9_]*)*)\}:\s*$/) { + if (defined $current) { + my @attachment_names = split /,/, $2; + ($attachment_rows, $attachment_fields) = ($1, scalar @attachment_names); + undef $attachment_rows if !$attachment_rows; + } else { $malformed++; } + next; + } + if ($line =~ /^ (.*)$/ && defined($attachment_rows)) { + my @values = csv_values($1); + $malformed++ if @values != $attachment_fields; + $attachment_rows--; + if (!$attachment_rows) { + undef $attachment_rows; + undef $attachment_fields; + } next; } + if ($line =~ /^\s*$/) { next; } + if ($line =~ /^\s/) { $malformed++; next; } last if $line =~ /^\S/; } + $malformed++ if defined($attachment_rows) && $attachment_rows; push @rows, $current if defined $current; close $fh; print encode_json({declared => $header ? 0 + $declared : 0, diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index 18446e3a17c..241f7cde0aa 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -3171,6 +3171,27 @@ EOF done pass "Lavish table and list forms preserve Unicode comments, answers, and reconcile notes" +cat > "$READ" <<'EOF' +session: + status: feedback +prompts[1]: + - uid: "el-a" + prompt "captain says stop" + selector: "section#comment" + tag: note + text: "Element text" +EOF +read_status=0 +out=$(read_out 2>&1) || read_status=$? +[ "$read_status" -ne 0 ] || fail "read certified a malformed list-form capture as complete" +assert_contains "$out" "presented_items: 1" \ + "a malformed list-form field hid the rest of its item" +assert_contains "$out" "malformed_items: 1" \ + "a malformed list-form field was not reported" +assert_contains "$out" "complete: no" \ + "a malformed list-form field was certified as complete" +pass "read never certifies malformed list-form fields as complete" + cat > "$READ" <<'EOF' session: file: /review.html diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index ef042f61885..2edeadad8ee 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -97,10 +97,18 @@ run_spawn() { FM_FAKE_LAUNCH_LOG="$launchlog" FM_FAKE_PI_VERSION="${FM_TEST_PI_VERSION:-0.84.0}" \ FM_FAKE_CURSOR_MODELS="${FM_TEST_CURSOR_MODELS:-}" \ FM_FAKE_CURSOR_LIST_STATUS="${FM_TEST_CURSOR_LIST_STATUS:-0}" \ + CODEX_HOME="${FM_TEST_CODEX_HOME:-$home/user-home/.codex}" \ GROK_HOME="$home/grok-home" \ fm_test_run_spawn "$home" "$wt" "$fakebin" "$@" } +seed_codex_catalog() { + local home=$1 contents=$2 codex_home + codex_home="$home/user-home/.codex" + mkdir -p "$codex_home" + printf '%s\n' "$contents" > "$codex_home/models_cache.json" +} + # Ship spawns carry an explicit delivery contract (AGENTS.md section 7); these # tests are about profile resolution, so they pass a fixed valid one. run_ship_spawn() { @@ -428,15 +436,17 @@ test_codex_threads_model_and_max_effort() { id=profile-codex-max-z4 rec=$(make_spawn_case profile-codex-max codex "$id") read_case_record "$rec" + seed_codex_catalog "$HOME_DIR" \ + '{"models":[{"slug":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' - out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --model gpt-5.6-luna --effort max) + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --model gpt-6-astra --effort max) status=$? - expect_code 0 "$status" "codex Luna spawn with max effort should succeed" - assert_meta_profile "$HOME_DIR/state/$id.meta" codex gpt-5.6-luna max + expect_code 0 "$status" "codex Astra spawn with max effort should succeed" + assert_meta_profile "$HOME_DIR/state/$id.meta" codex gpt-6-astra max launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "codex --model 'gpt-5.6-luna' -c 'model_reasoning_effort=\"max\"' --dangerously-bypass-approvals-and-sandbox" \ - "codex launch did not thread Luna's max reasoning effort config" - pass "codex Luna receives --model and model_reasoning_effort max profile flags" + assert_contains "$launch" "codex --model 'gpt-6-astra' -c 'model_reasoning_effort=\"max\"' --dangerously-bypass-approvals-and-sandbox" \ + "codex launch did not thread Astra's max reasoning effort config" + pass "codex Astra receives --model and model_reasoning_effort max profile flags" } test_codex_omits_max_effort_for_unsupported_model() { @@ -444,6 +454,8 @@ test_codex_omits_max_effort_for_unsupported_model() { id=profile-codex-max-unsupported-z4b rec=$(make_spawn_case profile-codex-max-unsupported codex "$id") read_case_record "$rec" + seed_codex_catalog "$HOME_DIR" \ + '{"models":[{"slug":"gpt-5","supported_reasoning_levels":[{"effort":"high"}]}]}' out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --model gpt-5 --effort max) status=$? From 10d8297e65f14b48611075ef896c39ea6410256b Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 03:14:24 +0800 Subject: [PATCH 09/21] no-mistakes(document): Align Codex and Lavish documentation --- bin/fm-spawn.sh | 4 ++-- docs/verification/process-event-sources.md | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 371986c9fd3..cd86e588ba6 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -2360,8 +2360,8 @@ effort_flag_for_harness() { esac ;; codex) - # Codex exposes model_reasoning_effort only for levels advertised by the - # selected model's installed catalog entry. + # Codex uses model_reasoning_effort for the verified shared levels. + # Max additionally requires support in the selected model's catalog entry. case "$effort" in low | medium | high | xhigh) printf -- '-c %s ' "$(shell_quote "model_reasoning_effort=\"$effort\"")" ;; max) diff --git a/docs/verification/process-event-sources.md b/docs/verification/process-event-sources.md index 50d7c60255e..4054855c2ec 100644 --- a/docs/verification/process-event-sources.md +++ b/docs/verification/process-event-sources.md @@ -104,7 +104,7 @@ Exercised by `tests/fm-procevent.test.sh` against a fake blocking source whose c | adapter-owned silence verdict | an ordinary firstmate-owned Lavish source driven against a stand-in poll that returns an empty ended session captures its result, records it durably handled, appends no wake, and stays silent through a later `reconcile` that would otherwise republish it, while still retiring its ended source; the same real path with a `Send & End` response carrying the captain's choice still publishes its `check` wake and is left unacknowledged for the handler | | worker-owned Lavish rounds | one three-round fixture arms a board for an identity-matched task endpoint, delivers nonterminal and terminal captures directly to that task's steering inbox without a firstmate `check` wake, acknowledges each nonterminal round through a successful re-arm, redelivers an inbox note filed before acknowledgement, refuses a second armer and every early retirement, and concludes the terminal round through `handled` without another poll; focused fixtures also pin failed re-arm rollback, generation-specific reply staging, one reply post across transient poll retries, unreachable-owner refusal, interrupted conclusion recovery, and repeat acknowledgement isolation | | Lavish handled-status classification | an executable fixture table pins exact `feedback`, `ended`, `waiting`, and `browser_disconnected` mappings, including `browser_disconnected` to `disconnected`; the same suite proves that status is nonterminal and receives a zero-answer silence verdict | -| Lavish result decoding | `tests/fm-procevent.test.sh` and `tests/fm-captain-hold-lifecycle.test.sh` use synthetic field-declared tables and YAML-like item lists to prove that `read` preserves every parsed row, `answers` and `reconciles` extract their respective choice rows from either form, and `silent` recognizes either representation as content; list coverage includes annotations, versioned choice context, nested attachment metadata, and a session-ending message; both representations preserve non-ASCII comments, answers, and reconcile notes; declared-count mismatches preserve parsed rows, report incomplete, and make `read` exit nonzero; a malformed table row also reports incomplete without hiding its valid neighbor | +| Lavish result decoding | `tests/fm-procevent.test.sh` and `tests/fm-captain-hold-lifecycle.test.sh` use synthetic field-declared tables and YAML-like item lists to prove that `read` preserves every parsed row, `answers` and `reconciles` extract their respective choice rows from either form, and `silent` recognizes either representation as content; list coverage includes annotations, versioned choice context, nested attachment metadata, and a session-ending message; both representations preserve non-ASCII comments, answers, and reconcile notes; declared-count mismatches preserve parsed rows, report incomplete, and make `read` exit nonzero; a malformed row in either representation also reports incomplete without hiding valid neighboring fields or rows | | session-derived Lavish routing | the three-round worker fixture starts its first listener under conflicting ambient host/port values and configuration, then recovers later listeners while that conflicting configuration remains, and proves every reply/poll uses the board's saved session endpoint; direct polls cover Unicode artifact paths, hostnames, IPv6, session endpoint changes, quiet retries, and refusal before reply consumption when session evidence is absent or invalid; spawn coverage still proves the configured opening address enters the worker launch | | silence fails closed | the adapter's published `silent` command suppresses only an `ended` session with no queued content block or a `browser_disconnected` response, and announces a real answer, freeform prose, any recognized content block regardless of its declared count, a malformed top-level content header, a `waiting` or `missing` session, a server error, an unreadable result, and indented payload text imitating an empty content block; the `remote-reply` and `when` adapters, which implement no `silent` command, announce every result | | terminal retirement preserves the result | the retired source's captured output, its announced event, its handled acknowledgement, and later explicit `retire` all still behave normally | From 48b50ae0d96726fdb44092e7acd6093a77f93747 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 03:18:45 +0800 Subject: [PATCH 10/21] no-mistakes(document): Document catalog-based Codex max effort --- docs/configuration.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/configuration.md b/docs/configuration.md index 587d2e4edd7..f08bcae0235 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -548,7 +548,7 @@ The resolver returns an actionable configuration error before any request when s A profile `floor` contains only `scope` and `min_percent`, always uses that profile's provider and matched account, and makes that one candidate ineligible below `min_percent` on the named scope. An absent or unknown named row also makes the candidate unrankable and is reported as an unverifiable floor, not as a known shortfall. `ultra` is native-only: the model-aware validation contract and launch mapping are owned by `bin/fm-harness.sh validate-native-effort` and `bin/fm-spawn.sh` respectively. -Codex `max` is valid when the profile selects `gpt-5.6-luna`, whose installed catalog entry supports that reasoning level. +For Codex `max`, the launch path passes the setting only when the selected model's entry in `${CODEX_HOME:-~/.codex}/models_cache.json` advertises that reasoning level; otherwise it warns and omits the setting. An omitted model or effort means the selected harness uses its own default for that axis. Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. From 61985d8086d3a7f009e4818a928711e924d5ec00 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 04:38:51 +0800 Subject: [PATCH 11/21] fix(pi): support Pi 1.0 rendering contracts --- .pi/extensions/fm-branch-supervision.ts | 25 ++++++++++++++++++++----- tests/fm-calm-pi-extension.test.sh | 12 ++++++++---- 2 files changed, 28 insertions(+), 9 deletions(-) diff --git a/.pi/extensions/fm-branch-supervision.ts b/.pi/extensions/fm-branch-supervision.ts index 74ccac0be9d..a5a89ab57b8 100644 --- a/.pi/extensions/fm-branch-supervision.ts +++ b/.pi/extensions/fm-branch-supervision.ts @@ -2097,12 +2097,14 @@ ${context.command} }; let stockOutcomesPreviewLines: number | null | undefined; - const getStockOutcomesPreviewLines = (): number | undefined => { - if (stockOutcomesPreviewLines !== undefined) return stockOutcomesPreviewLines ?? undefined; + let stockOutcomesCallShowsArgs: boolean | undefined; + const probeStockOutcomesRendering = (): void => { + if (stockOutcomesPreviewLines !== undefined && stockOutcomesCallShowsArgs !== undefined) return; const probeTokens = Array.from( { length: 64 }, (_, index) => `FM_OUTCOMES_PREVIEW_PROBE_${String(index).padStart(2, "0")}`, ); + const callArgProbe = "FM_OUTCOMES_CALL_ARGS_PROBE"; try { const probeDefinition: ToolDefinition = { name: "fm_outcomes_preview_probe", @@ -2114,7 +2116,7 @@ ${context.command} const probe = new ToolExecutionComponent( probeDefinition.name, "fm-outcomes-preview-probe", - {}, + { probe: callArgProbe }, { showImages: false }, probeDefinition, { requestRender() {} } as ConstructorParameters[5], @@ -2127,9 +2129,14 @@ ${context.command} const rendered = probe.render(4096).join("\n"); const visibleLines = probeTokens.filter((token) => rendered.includes(token)).length; stockOutcomesPreviewLines = visibleLines > 0 && visibleLines < probeTokens.length ? visibleLines : null; + stockOutcomesCallShowsArgs = rendered.includes(callArgProbe); } catch { stockOutcomesPreviewLines = null; + stockOutcomesCallShowsArgs = false; } + }; + const getStockOutcomesPreviewLines = (): number | undefined => { + probeStockOutcomesRendering(); return stockOutcomesPreviewLines ?? undefined; }; @@ -2167,11 +2174,19 @@ ${context.command} recent: Type.Optional(Type.Number({ description: "How many most-recent outcomes to read (default 20)" })), }), renderShell: "self", - renderCall: (_args, theme, context) => { + renderCall: (args, theme, context) => { if (calmPresentation.stockExportRendering) throw new Error("Use Pi stock export rendering"); if (calmHides("assistant-tool-call")) return new Container(); const shellState = context.state as OutcomesToolShellState; - shellState.call = new Text(theme.fg("toolTitle", theme.bold("fm_branch_outcomes")), 0, 0); + probeStockOutcomesRendering(); + let call = theme.fg("toolTitle", theme.bold("fm_branch_outcomes")); + const recent = (args as { recent?: unknown }).recent; + if (stockOutcomesCallShowsArgs && recent !== undefined) { + call += context.expanded + ? `\n${theme.fg("muted", ` recent: ${JSON.stringify(recent)}`)}` + : ` ${theme.fg("muted", `recent=${JSON.stringify(recent)}`)}`; + } + shellState.call = new Text(call, 0, 0); return refreshOutcomesToolShell(shellState, theme, context); }, renderResult: (result, options, theme, context) => { diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index 02cee20e6e3..899fc5fb675 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -3388,13 +3388,17 @@ SH } test_interactive_terminal_e2e() { - local project config home session_file export_file export_dom default_snapshot expanded_snapshot hidden_snapshot active_before_snapshot active_hidden_snapshot export_snapshot export_settled_snapshot restored_snapshot working_snapshot working_response_snapshot restarted_snapshot resumed_restored_snapshot hash_before hash_after now version chrome chrome_report active_wait active_screen_wait boat_frame_one boat_frame_two boat_resized_snapshot boat_focus_snapshot boat_cleared_snapshot boat_hull_line boat_sail_line boat_column_one boat_column_two boat_line boat_color_snapshot boat_color_line boat_water_snapshot boat_water_line boat_water_first boat_water_changed boat_narrow_snapshot boat_freeze_snapshot boat_resume_snapshot boat_freeze_column boat_freeze_sail boat_resume_column boat_resume_sail + local project config home session_file export_file export_dom default_snapshot expanded_snapshot hidden_snapshot active_before_snapshot active_hidden_snapshot export_snapshot export_settled_snapshot restored_snapshot working_snapshot working_response_snapshot restarted_snapshot resumed_restored_snapshot hash_before hash_after now version chrome chrome_report active_wait active_screen_wait boat_frame_one boat_frame_two boat_resized_snapshot boat_focus_snapshot boat_cleared_snapshot boat_hull_line boat_sail_line boat_column_one boat_column_two boat_line boat_color_snapshot boat_color_line boat_water_snapshot boat_water_line boat_water_first boat_water_changed boat_narrow_snapshot boat_freeze_snapshot boat_resume_snapshot boat_freeze_column boat_freeze_sail boat_resume_column boat_resume_sail tui_mode_args if ! command -v pi >/dev/null 2>&1 || ! command -v tmux >/dev/null 2>&1; then echo "skip: pi or tmux not found for Pi calm interactive E2E" return 0 fi version=$(pi --version 2>/dev/null || true) record_pi_version_evidence "$version" "Pi calm interactive E2E" + tui_mode_args= + if pi --help 2>&1 | grep -Fq -- '--tui-mode'; then + tui_mode_args='--tui-mode regular' + fi project="$TMP_ROOT/e2e-project" config="$TMP_ROOT/e2e-config" @@ -3634,7 +3638,7 @@ TS JSON tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 180 -y 44 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi $tui_mode_args --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" wait_for_text "$default_snapshot" "The deterministic tool example is complete." \ || fail "Pi calm E2E did not reach the restored session transcript" assert_contains "$(cat "$default_snapshot")" "CALM_E2E_OUTPUT" "calm mode was not off by default" @@ -3849,7 +3853,7 @@ if (!messages || !tree) process.exit(1); if (!/
]*>[\s\S]*Show a deterministic tool example\./.test(messages)) process.exit(1); if (!/
]*>[\s\S]*The deterministic tool example is complete\./.test(messages)) process.exit(1); if (messages.includes('
]*>[\s\S]*\[firstmate-synthetic-input\] · Hidden in terminal/.test(messages)) process.exit(1); for (const current of ["CURRENT_WATCHER_E2E", "CURRENT_TURN_END_E2E", "CURRENT_AWAY_E2E", "CURRENT_FROM_FIRSTMATE_E2E", "CURRENT_LAUNCH_BRIEF_E2E"]) { if (!messages.includes(current)) process.exit(1); } @@ -4247,7 +4251,7 @@ JS tmux -L "$TMUX_SOCKET" kill-session -t "$TMUX_SESSION" 2>/dev/null || true tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 180 -y 44 \ - "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' PI_OFFLINE=1 pi $tui_mode_args --approve --no-skills --no-prompt-templates --no-context-files --session '$session_file'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 30" wait_for_text "$restarted_snapshot" "CALM_WORKING_E2E_RESPONSE" \ || fail "Pi did not restore the persisted session after restart" assert_not_contains "$(cat "$restarted_snapshot")" "CALM_E2E_OUTPUT" "restart/resume reset Calm and restored a tool row" From 42999894e57fbde4d769958d398eaf92014cc729 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 05:13:48 +0800 Subject: [PATCH 12/21] Validate Codex max effort from catalog --- bin/fm-bootstrap.sh | 13 +++++++++++-- bin/fm-codex-catalog-lib.sh | 35 +++++++++++++++++++++++++++++++++++ bin/fm-dispatch-resolve.sh | 13 +++++++++++-- bin/fm-spawn.sh | 19 ++++--------------- tests/fm-bootstrap.test.sh | 11 ++++++++--- 5 files changed, 69 insertions(+), 22 deletions(-) create mode 100755 bin/fm-codex-catalog-lib.sh diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index 9223cbf0a85..c395d781468 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -185,6 +185,8 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-control-lib.sh" # shellcheck source=bin/fm-env-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-env-lib.sh" +# shellcheck source=bin/fm-codex-catalog-lib.sh disable=SC1091 +. "$SCRIPT_DIR/fm-codex-catalog-lib.sh" # shellcheck source=bin/fm-tangle-lib.sh disable=SC1091 . "$SCRIPT_DIR/fm-tangle-lib.sh" # shellcheck source=bin/fm-ff-lib.sh disable=SC1091 @@ -1137,7 +1139,14 @@ crew_dispatch_validate() { else verified_harnesses='["claude","codex","opencode","pi","pi-signed","grok","kimi","cursor","agy","muse","rovo","omp"]' fi - err=$(jq -r --argjson typed "$typed_active" --argjson verified_harnesses "$verified_harnesses" --arg provider_re "$FM_QUOTA_PROVIDER_ID_RE" ' + codex_max_models='[]' + if codex_max_models=$(fm_codex_catalog_models_supporting_effort max | jq -Rsc 'split("\n") | map(select(length > 0))'); then + : + else + codex_max_models='[]' + fi + err=$(jq -r --argjson typed "$typed_active" --argjson verified_harnesses "$verified_harnesses" \ + --argjson codex_max_models "$codex_max_models" --arg provider_re "$FM_QUOTA_PROVIDER_ID_RE" ' def verified($h): $verified_harnesses | index($h); def provider_id($p): ($p | type) == "string" and ($p | test($provider_re)); def effort_ok($h; $m; $e): @@ -1145,7 +1154,7 @@ crew_dispatch_validate() { elif ($e | type) != "string" then false elif $e == "ultra" then (($h == "pi" or $h == "pi-signed") and (($m | type) == "string") and ($m | startswith("codex-native/")) and ($m | length) > 13) elif $h == "claude" then (["low","medium","high","xhigh","max"] | index($e)) - elif $h == "codex" then ((["low","medium","high","xhigh"] | index($e)) != null or ($e == "max" and $m == "gpt-5.6-luna")) + elif $h == "codex" then ((["low","medium","high","xhigh"] | index($e)) != null or ($e == "max" and ($codex_max_models | index($m)) != null)) elif $h == "grok" then (["low","medium","high"] | index($e)) elif $h == "agy" then (["low","medium","high"] | index($e)) elif $h == "pi" or $h == "pi-signed" or $h == "omp" then (["low","medium","high","xhigh","max"] | index($e)) diff --git a/bin/fm-codex-catalog-lib.sh b/bin/fm-codex-catalog-lib.sh new file mode 100755 index 00000000000..c64a7132ae0 --- /dev/null +++ b/bin/fm-codex-catalog-lib.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash + +fm_codex_catalog_path() { + printf '%s\n' "${CODEX_HOME:-$HOME/.codex}/models_cache.json" +} + +fm_codex_catalog_supports_effort() { + local model=$1 effort=$2 catalog + [ -n "$model" ] && [ "$model" != default ] || return 1 + command -v jq >/dev/null 2>&1 || return 1 + catalog=$(fm_codex_catalog_path) + jq -e --arg model "$model" --arg effort "$effort" ' + any(.models[]?; + ((.slug? // .id? // .model?) == $model) + and any(.supported_reasoning_levels[]?; .effort? == $effort) + ) + ' "$catalog" >/dev/null 2>&1 +} + +fm_codex_catalog_models_supporting_effort() { + local effort=$1 catalog + command -v jq >/dev/null 2>&1 || return 1 + catalog=$(fm_codex_catalog_path) + jq -r --arg effort "$effort" ' + .models[]? + | select(any(.supported_reasoning_levels[]?; .effort? == $effort)) + | (.slug? // .id? // .model?) + | select(type == "string" and length > 0) + ' "$catalog" 2>/dev/null +} + +fm_codex_catalog_warn_dropped_effort() { + printf 'warning: dropped codex effort %s for model %s; catalog does not advertise it\n' \ + "$1" "${2:-default}" >&2 +} diff --git a/bin/fm-dispatch-resolve.sh b/bin/fm-dispatch-resolve.sh index 3dac9d143ef..beda72fb02f 100755 --- a/bin/fm-dispatch-resolve.sh +++ b/bin/fm-dispatch-resolve.sh @@ -70,6 +70,8 @@ CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" . "$SCRIPT_DIR/fm-control-lib.sh" # shellcheck source=bin/fm-env-lib.sh . "$SCRIPT_DIR/fm-env-lib.sh" +# shellcheck source=bin/fm-codex-catalog-lib.sh +. "$SCRIPT_DIR/fm-codex-catalog-lib.sh" # shellcheck source=bin/fm-timing-lib.sh . "$SCRIPT_DIR/fm-timing-lib.sh" @@ -122,10 +124,17 @@ trap 'rm -f "$RULES"' EXIT cp "$RULES_PATH" "$RULES" || die "could not snapshot rules file: $RULES_PATH" chmod 400 "$RULES" || die "could not protect rules snapshot" VERIFIED_HARNESSES=$(fm_control_harnesses | jq -Rsc 'split("\n") | map(select(length > 0))') +CODEX_MAX_MODELS='[]' +if CODEX_MAX_MODELS=$(fm_codex_catalog_models_supporting_effort max | jq -Rsc 'split("\n") | map(select(length > 0))'); then + : +else + CODEX_MAX_MODELS='[]' +fi # The fields this tool consumes must be well formed; bootstrap owns the wider # schema diagnostic, but an intake never selects around a malformed file. -rules_err=$(jq -r --argjson verified_harnesses "$VERIFIED_HARNESSES" --arg provider_re "$FM_QUOTA_PROVIDER_ID_RE" ' +rules_err=$(jq -r --argjson verified_harnesses "$VERIFIED_HARNESSES" \ + --argjson codex_max_models "$CODEX_MAX_MODELS" --arg provider_re "$FM_QUOTA_PROVIDER_ID_RE" ' def verified($h): $verified_harnesses | index($h); def provider_id($p): ($p | type) == "string" and ($p | test($provider_re)); def effort_ok($h; $m; $e): @@ -133,7 +142,7 @@ rules_err=$(jq -r --argjson verified_harnesses "$VERIFIED_HARNESSES" --arg provi elif ($e | type) != "string" then false elif $e == "ultra" then (($h == "pi" or $h == "pi-signed") and (($m | type) == "string") and ($m | startswith("codex-native/")) and ($m | length) > 13) elif $h == "claude" then (["low","medium","high","xhigh","max"] | index($e)) != null - elif $h == "codex" then ((["low","medium","high","xhigh"] | index($e)) != null or ($e == "max" and $m == "gpt-5.6-luna")) + elif $h == "codex" then ((["low","medium","high","xhigh"] | index($e)) != null or ($e == "max" and ($codex_max_models | index($m)) != null)) elif $h == "grok" or $h == "agy" then (["low","medium","high"] | index($e)) != null elif $h == "pi" or $h == "pi-signed" or $h == "omp" or $h == "muse" then (["low","medium","high","xhigh","max"] | index($e)) != null elif $h == "rovo" then (["low","medium","high","max"] | index($e)) != null diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index cd86e588ba6..c6086b5c249 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -475,6 +475,8 @@ PROJECTS="${FM_PROJECTS_OVERRIDE:-$FM_HOME/projects}" CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" # shellcheck source=bin/fm-config-inherit-lib.sh . "$SCRIPT_DIR/fm-config-inherit-lib.sh" +# shellcheck source=bin/fm-codex-catalog-lib.sh +. "$SCRIPT_DIR/fm-codex-catalog-lib.sh" if ! LAUNCH_ENV_ENABLED=$(fm_config_source_present "$CONFIG/launch-env-allowlist"); then exit 1 fi @@ -2338,18 +2340,6 @@ model_flag_for_harness() { esac } -codex_catalog_supports_effort() { - local model=$1 effort=$2 catalog="${CODEX_HOME:-$HOME/.codex}/models_cache.json" - [ -n "$model" ] && [ "$model" != default ] || return 1 - command -v jq >/dev/null 2>&1 || return 1 - jq -e --arg model "$model" --arg effort "$effort" ' - any(.models[]?; - ((.slug? // .id? // .model?) == $model) - and any(.supported_reasoning_levels[]?; .effort? == $effort) - ) - ' "$catalog" >/dev/null 2>&1 -} - effort_flag_for_harness() { local harness=$1 effort=$2 model=${3:-} [ -n "$effort" ] && [ "$effort" != default ] || return 0 @@ -2365,11 +2355,10 @@ effort_flag_for_harness() { case "$effort" in low | medium | high | xhigh) printf -- '-c %s ' "$(shell_quote "model_reasoning_effort=\"$effort\"")" ;; max) - if codex_catalog_supports_effort "$model" "$effort"; then + if fm_codex_catalog_supports_effort "$model" "$effort"; then printf -- '-c %s ' "$(shell_quote 'model_reasoning_effort="max"')" else - printf 'warning: dropped codex effort %s for model %s; catalog does not advertise it\n' \ - "$effort" "${model:-default}" >&2 + fm_codex_catalog_warn_dropped_effort "$effort" "$model" fi ;; esac diff --git a/tests/fm-bootstrap.test.sh b/tests/fm-bootstrap.test.sh index dc24b267399..04fe756f83d 100755 --- a/tests/fm-bootstrap.test.sh +++ b/tests/fm-bootstrap.test.sh @@ -1115,13 +1115,17 @@ test_crew_dispatch_validation() { [ -n "$label" ] || continue n=$((n + 1)) case_dir="$TMP_ROOT/dispatch-$n" - mkdir -p "$case_dir/home/config" + mkdir -p "$case_dir/home/config" "$case_dir/codex" printf '%s\n' manual > "$case_dir/home/config/backlog-backend" + cat > "$case_dir/codex/models_cache.json" <<'JSON' +{"models":[{"slug":"gpt-5.6-luna","supported_reasoning_levels":[{"effort":"max"}]},{"slug":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]} +JSON printf '%s\n' "$body" > "$case_dir/home/config/crew-dispatch.json" fakebin=$(make_fake_toolchain "$case_dir") add_real_jq "$fakebin" - out=$(PATH="$fakebin:$BASE_PATH" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ - TYPESAFE_API_KEY=test-key FM_FAKE_TREEHOUSE_LEASE_HELP=1 "$ROOT/bin/fm-bootstrap.sh") + out=$(PATH="$fakebin:$BASE_PATH" CODEX_HOME="$case_dir/codex" FM_HOME="$case_dir/home" \ + FM_ROOT_OVERRIDE="$case_dir/home" TYPESAFE_API_KEY=test-key \ + FM_FAKE_TREEHOUSE_LEASE_HELP=1 "$ROOT/bin/fm-bootstrap.sh") case "$mode" in empty) [ -z "$out" ] || fail "$label: expected silence, got: $out" ;; @@ -1134,6 +1138,7 @@ test_crew_dispatch_validation() { malformed dispatch config is flagged^{"rules":[^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - malformed JSON unverified dispatch harness is flagged^{"rules":[{"when":"anything","use":{"harness":"spaceship"}}],"default":{"harness":"codex"}}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - unverified harness: spaceship codex Luna max effort is accepted^{"rules":[{"when":"big feature","use":{"harness":"codex","model":"gpt-5.6-luna","effort":"max"}}]}^empty^ +codex Astra max effort is accepted from catalog^{"rules":[{"when":"big feature","use":{"harness":"codex","model":"gpt-6-astra","effort":"max"}}]}^empty^ codex unsupported model max effort is flagged^{"rules":[{"when":"big feature","use":{"harness":"codex","model":"gpt-5","effort":"max"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: codex:max unsupported grok max effort is flagged^{"rules":[{"when":"deep current work","use":{"harness":"grok","model":"grok-4","effort":"max"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: grok:max unsupported grok xhigh effort is flagged^{"rules":[{"when":"deep current work","use":{"harness":"grok","model":"grok-4","effort":"xhigh"}}]}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: grok:xhigh From 25ab3efd945d49be6e53c5a8a793b654f324924c Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 05:49:43 +0800 Subject: [PATCH 13/21] no-mistakes(review): Preserve Lavish metadata and harden Codex catalog validation --- bin/fm-codex-catalog-lib.sh | 11 ++-- bin/fm-procevent-lavish.sh | 62 ++++++++++++++++++--- tests/fm-bootstrap.test.sh | 27 +++++++++ tests/fm-dispatch-resolve.test.sh | 25 +++++++++ tests/fm-procevent.test.sh | 92 +++++++++++++++++++++++++++++++ 5 files changed, 204 insertions(+), 13 deletions(-) diff --git a/bin/fm-codex-catalog-lib.sh b/bin/fm-codex-catalog-lib.sh index c64a7132ae0..2feb6694c95 100755 --- a/bin/fm-codex-catalog-lib.sh +++ b/bin/fm-codex-catalog-lib.sh @@ -11,22 +11,23 @@ fm_codex_catalog_supports_effort() { catalog=$(fm_codex_catalog_path) jq -e --arg model "$model" --arg effort "$effort" ' any(.models[]?; - ((.slug? // .id? // .model?) == $model) + (.slug? == $model) and any(.supported_reasoning_levels[]?; .effort? == $effort) ) ' "$catalog" >/dev/null 2>&1 } fm_codex_catalog_models_supporting_effort() { - local effort=$1 catalog + local effort=$1 catalog models command -v jq >/dev/null 2>&1 || return 1 catalog=$(fm_codex_catalog_path) - jq -r --arg effort "$effort" ' + models=$(jq -r --arg effort "$effort" ' .models[]? | select(any(.supported_reasoning_levels[]?; .effort? == $effort)) - | (.slug? // .id? // .model?) + | .slug? | select(type == "string" and length > 0) - ' "$catalog" 2>/dev/null + ' "$catalog" 2>/dev/null) || return 1 + [ -z "$models" ] || printf '%s\n' "$models" } fm_codex_catalog_warn_dropped_effort() { diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 2a846213367..9c9a3403b96 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -561,7 +561,8 @@ prompt_rows_json() { # } return @values; } - my ($current, $line, $attachment_rows, $attachment_fields); + my ($current, $line, $target_lines, $attachment_rows); + my @attachment_fields; while (defined($line = <$fh>)) { if (!$header) { if ($line =~ /^(?:prompts|feedback)\[(\d+)\]\{([^}]*)\}:\s*$/) { @@ -595,7 +596,17 @@ prompt_rows_json() { # if (defined($attachment_rows) && $line !~ /^ /) { $malformed++ if $attachment_rows; undef $attachment_rows; - undef $attachment_fields; + @attachment_fields = (); + } + if (defined($target_lines) && $line =~ /^ (.*)$/) { + my $target_line = $1; + chomp $target_line; + push @$target_lines, $target_line; + next; + } + if (defined($target_lines)) { + $malformed++ unless @$target_lines; + undef $target_lines; } if ($line =~ /^ -\s+(.+)$/) { push @rows, $current if defined $current; @@ -605,35 +616,50 @@ prompt_rows_json() { # } else { $malformed++; } next; } - if ($line =~ /^ ([A-Za-z_][A-Za-z0-9_]*):\s*(.*)$/) { + if ($line =~ /^ target:\s*$/) { if (defined $current) { - $current->{$1} = unquote($2); + $current->{target} = []; + $target_lines = $current->{target}; } else { $malformed++; } next; } if ($line =~ /^ attachments\[(\d+)\]\{([A-Za-z_][A-Za-z0-9_]*(?:,[A-Za-z_][A-Za-z0-9_]*)*)\}:\s*$/) { if (defined $current) { - my @attachment_names = split /,/, $2; - ($attachment_rows, $attachment_fields) = ($1, scalar @attachment_names); + @attachment_fields = split /,/, $2; + $current->{attachments} = []; + $attachment_rows = $1; undef $attachment_rows if !$attachment_rows; } else { $malformed++; } next; } if ($line =~ /^ (.*)$/ && defined($attachment_rows)) { my @values = csv_values($1); - $malformed++ if @values != $attachment_fields; + if (@values == @attachment_fields) { + my %attachment; + $attachment{$attachment_fields[$_]} = $values[$_] for 0 .. $#attachment_fields; + push @{$current->{attachments}}, \%attachment; + } else { + $malformed++; + } $attachment_rows--; if (!$attachment_rows) { undef $attachment_rows; - undef $attachment_fields; + @attachment_fields = (); } next; } + if ($line =~ /^ ([A-Za-z_][A-Za-z0-9_]*):\s*(.*)$/) { + if (defined $current) { + $current->{$1} = unquote($2); + } else { $malformed++; } + next; + } if ($line =~ /^\s*$/) { next; } if ($line =~ /^\s/) { $malformed++; next; } last if $line =~ /^\S/; } $malformed++ if defined($attachment_rows) && $attachment_rows; + $malformed++ if defined($target_lines) && !@$target_lines; push @rows, $current if defined $current; close $fh; print encode_json({declared => $header ? 0 + $declared : 0, @@ -790,6 +816,24 @@ cmd_read() { return if !@lines || (@lines == 1 && $lines[0] eq ""); print "| $_\n" for @lines; } + sub emit_prompt_metadata { + my ($prompt) = @_; + if (ref($prompt->{target}) eq "ARRAY" && @{$prompt->{target}}) { + print "target:\n"; + emit_body(join("\n", @{$prompt->{target}})); + } + if (ref($prompt->{attachments}) eq "ARRAY" && @{$prompt->{attachments}}) { + print "attachment_count: ", scalar(@{$prompt->{attachments}}), "\n"; + for my $attachment_index (0 .. $#{$prompt->{attachments}}) { + my $attachment = $prompt->{attachments}[$attachment_index]; + print "ATTACHMENT ", ($attachment_index + 1), " of ", scalar(@{$prompt->{attachments}}), "\n"; + for my $key (sort keys %$attachment) { + print "attachment_$key:\n"; + emit_body($attachment->{$key}); + } + } + } + } if (@messages) { my $message_label = $session_ended =~ /^(?:true|True|TRUE)$/ ? "SESSION-ENDING MESSAGE" : "CAPTAIN MESSAGE"; @@ -800,6 +844,7 @@ cmd_read() { ? $messages[$i]{prompt} : (defined $messages[$i]{text} ? $messages[$i]{text} : ""); emit_body($body); + emit_prompt_metadata($messages[$i]); } print "END $message_label\n"; } else { @@ -836,6 +881,7 @@ cmd_read() { print "prompt:\n"; emit_body($comment); } + emit_prompt_metadata($f); } print "END ANNOTATIONS\n"; } else { diff --git a/tests/fm-bootstrap.test.sh b/tests/fm-bootstrap.test.sh index 04fe756f83d..253bc52d687 100755 --- a/tests/fm-bootstrap.test.sh +++ b/tests/fm-bootstrap.test.sh @@ -1193,6 +1193,33 @@ default profile floor without min_percent is flagged^{"default":[{"harness":"cod default profile floor provider override is flagged^{"default":{"harness":"codex","floor":{"scope":"all_models","min_percent":50,"provider":"claude"}}}^exact^CREW_DISPATCH: invalid config/crew-dispatch.json - default profile floor needs scope and min_percent 0..100 ROWS + case_dir="$TMP_ROOT/dispatch-catalog-integrity" + mkdir -p "$case_dir/home/config" "$case_dir/codex" + printf '%s\n' manual > "$case_dir/home/config/backlog-backend" + printf '%s\n' '{"rules":[{"when":"big feature","use":{"harness":"codex","model":"gpt-6-astra","effort":"max"}}]}' \ + > "$case_dir/home/config/crew-dispatch.json" + fakebin=$(make_fake_toolchain "$case_dir") + add_real_jq "$fakebin" + + printf '%s\n' \ + '{"models":[{"slug":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ + '{' > "$case_dir/codex/models_cache.json" + out=$(PATH="$fakebin:$BASE_PATH" CODEX_HOME="$case_dir/codex" FM_HOME="$case_dir/home" \ + FM_ROOT_OVERRIDE="$case_dir/home" TYPESAFE_API_KEY=test-key \ + FM_FAKE_TREEHOUSE_LEASE_HELP=1 "$ROOT/bin/fm-bootstrap.sh") + [ "$out" = 'CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: codex:max' ] \ + || fail "malformed catalog data partially authorized Codex max, got: $out" + + printf '%s\n' \ + '{"models":[{"slug":"other-model","id":"gpt-6-astra","model":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ + > "$case_dir/codex/models_cache.json" + out=$(PATH="$fakebin:$BASE_PATH" CODEX_HOME="$case_dir/codex" FM_HOME="$case_dir/home" \ + FM_ROOT_OVERRIDE="$case_dir/home" TYPESAFE_API_KEY=test-key \ + FM_FAKE_TREEHOUSE_LEASE_HELP=1 "$ROOT/bin/fm-bootstrap.sh") + [ "$out" = 'CREW_DISPATCH: invalid config/crew-dispatch.json - invalid effort: codex:max' ] \ + || fail "non-slug catalog aliases authorized Codex max, got: $out" + pass "bootstrap accepts Codex max only from a complete slug-matched catalog" + case_dir="$TMP_ROOT/dispatch-opt-in-gate" mkdir -p "$case_dir/home/config" printf '%s\n' manual > "$case_dir/home/config/backlog-backend" diff --git a/tests/fm-dispatch-resolve.test.sh b/tests/fm-dispatch-resolve.test.sh index 0524d190501..41f10c77374 100755 --- a/tests/fm-dispatch-resolve.test.sh +++ b/tests/fm-dispatch-resolve.test.sh @@ -727,6 +727,31 @@ printf '%s\n' '{"rules":[' > "$RULES" TYPESAFE_API_KEY=$KEY run code out err "$BRIEF" expect_code 2 "$code" "non-JSON rules exits 2" assert_contains "$err" 'not JSON' "non-JSON rules is named" + +CODEX_CATALOG="$TMP_ROOT/codex-catalog" +mkdir -p "$CODEX_CATALOG" +printf '%s\n' '{"rules":[{"when":"x","use":{"harness":"codex","model":"gpt-6-astra","effort":"max"}}]}' > "$RULES" +printf '%s\n' \ + '{"models":[{"slug":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ + '{' > "$CODEX_CATALOG/models_cache.json" +reset_log +CODEX_HOME="$CODEX_CATALOG" TYPESAFE_API_KEY=$KEY run code out err "$BRIEF" +expect_code 2 "$code" "a partially malformed catalog rejects Codex max" +assert_contains "$err" 'each use profile effort must be supported by its harness and model' \ + "partial catalog output authorized Codex max" +assert_absent "$LOG/argv" "a malformed catalog reached dispatch resolution" + +printf '%s\n' \ + '{"models":[{"slug":"other-model","id":"gpt-6-astra","model":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ + > "$CODEX_CATALOG/models_cache.json" +reset_log +CODEX_HOME="$CODEX_CATALOG" TYPESAFE_API_KEY=$KEY run code out err "$BRIEF" +expect_code 2 "$code" "non-slug aliases reject Codex max" +assert_contains "$err" 'each use profile effort must be supported by its harness and model' \ + "a non-slug catalog alias authorized Codex max" +assert_absent "$LOG/argv" "a non-slug catalog alias reached dispatch resolution" +pass "dispatch validation accepts only complete slug-matched Codex catalogs" + for bad in \ '{"rules":[{"when":"x","use":{"harness":"claude"},"approval":"firstmate"}]}|approval must be "captain" when present' \ '{"rules":[{"when":"x","use":{"harness":"claude"},"select":"mystery"}]}|unknown select: mystery' \ diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index 241f7cde0aa..b83e6ee8469 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -3171,6 +3171,98 @@ EOF done pass "Lavish table and list forms preserve Unicode comments, answers, and reconcile notes" +cat > "$READ" <<'EOF' +session: + status: feedback +prompts[6]: + - uid: "text-a" + prompt: "Change this phrase" + selector: "#intro" + tag: text + text: "Selected phrase" + target: + type: text-range + text: "Selected phrase" + selector: "#intro" + start: + selector: "#intro" + path[2]: 0,1 + offset: 2 + end: + selector: "#intro" + path[2]: 0,1 + offset: 17 + - uid: "cell-a" + prompt: "Update the value" + selector: "#plans td" + tag: td + text: "$20" + target: + type: table-cell + selector: "#plans td" + rowLabel: Pro + columnLabel: Price + text: "$20" + - uid: "node-a" + prompt: "Rename this node" + selector: "#flow g.node" + tag: mermaid-node + text: Queue + target: + type: mermaid-node + diagramId: flow + nodeId: queue + label: Queue + selector: "#flow g.node" + - uid: "whiteboard-a" + prompt: "Moved two nodes" + selector: "" + tag: whiteboard + text: "Diagram 1" + target: + type: excalidraw-scene + diagramIndex: 0 + scenePath: /tmp/review/0.excalidraw + previewPath: /tmp/review/0.png + stats: + added: 0 + moved: 2 + - uid: "layout-a" + prompt: "Fix the overflow" + selector: "" + tag: layout-warnings + text: "Layout issue: 1 selected" + target: + type: layout-warnings + artifact_revision: 4 + warnings[1]{id,rule,selector,component,axis,overflow_px,viewport_class,viewport_width,status,last_seen_at}: + warn-a,viewport-overflow,main,.card,horizontal,12,mobile,390,active,2030-01-01T00:00:00Z + - uid: "" + prompt: "See the attached reference" + selector: "" + tag: message + text: "" + attachments[1]{id,type,path,mime,bytes,width,height}: + image-a,image,/tmp/review/reference.png,image/png,1234,800,600 +EOF +out=$(read_out) || fail "read rejected valid target and attachment metadata" +assert_contains "$out" "presented_items: 6" "target-bearing prompts were dropped" +assert_contains "$out" "malformed_items: 0" "valid nested target metadata was marked malformed" +assert_contains "$out" "| path[2]: 0,1" "text-range path metadata was dropped" +assert_contains "$out" "| rowLabel: Pro" "table-cell target metadata was dropped" +assert_contains "$out" "| nodeId: queue" "Mermaid target metadata was dropped" +assert_contains "$out" "| scenePath: /tmp/review/0.excalidraw" "whiteboard scene path was dropped" +assert_contains "$out" "| previewPath: /tmp/review/0.png" "whiteboard preview path was dropped" +assert_contains "$out" "| warnings[1]{id,rule,selector,component,axis,overflow_px,viewport_class,viewport_width,status,last_seen_at}:" \ + "layout-warning target metadata was dropped" +assert_contains "$out" "attachment_path:" "attachment path label was dropped" +assert_contains "$out" "| /tmp/review/reference.png" "attachment path was dropped" +assert_contains "$out" "attachment_mime:" "attachment MIME label was dropped" +assert_contains "$out" "| image/png" "attachment MIME was dropped" +assert_contains "$out" "attachment_width:" "attachment dimensions were dropped" +assert_contains "$out" "| 800" "attachment width was dropped" +pass "read preserves Lavish targets and message attachment metadata" + cat > "$READ" <<'EOF' session: status: feedback From 529863755b6a86549e46be1c26f62e5a70e58903 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:10:02 +0800 Subject: [PATCH 14/21] no-mistakes(review): Relay Codex max downgrade warnings through recovery --- bin/fm-bootstrap.sh | 2 ++ bin/fm-codex-catalog-lib.sh | 11 ++++++++++ bin/fm-remote-secondmate-control.sh | 3 +++ bin/fm-spawn.sh | 1 + tests/fm-secondmate-liveness.test.sh | 21 ++++++++++++++++++ tests/fm-secondmate-sync.test.sh | 32 ++++++++++++++++++++++++++++ 6 files changed, 70 insertions(+) diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index c395d781468..5d75cc68811 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -802,6 +802,7 @@ secondmate_liveness_one() { # dead|missing) cause="remote endpoint $agent_state on its configured host" if out=$(FM_SPAWN_NO_GUARD=1 "$FM_ROOT/bin/fm-spawn.sh" "$id" --secondmate 2>&1); then + fm_codex_catalog_relay_dropped_effort_warnings "$out" secondmate_note_respawned "$id" report_relaunch "$id" "$cause" "host=$remote_host" else @@ -839,6 +840,7 @@ secondmate_liveness_one() { # cause="recorded endpoint confidently missing" fi if out=$(FM_SPAWN_NO_GUARD=1 "$FM_ROOT/bin/fm-spawn.sh" "$id" --secondmate 2>&1); then + fm_codex_catalog_relay_dropped_effort_warnings "$out" secondmate_note_respawned "$id" report_relaunch "$id" "$cause" "backend=$backend" else diff --git a/bin/fm-codex-catalog-lib.sh b/bin/fm-codex-catalog-lib.sh index 2feb6694c95..9b017290870 100755 --- a/bin/fm-codex-catalog-lib.sh +++ b/bin/fm-codex-catalog-lib.sh @@ -34,3 +34,14 @@ fm_codex_catalog_warn_dropped_effort() { printf 'warning: dropped codex effort %s for model %s; catalog does not advertise it\n' \ "$1" "${2:-default}" >&2 } + +fm_codex_catalog_relay_dropped_effort_warnings() { + local output=$1 line + while IFS= read -r line; do + case "$line" in + warning:\ dropped\ codex\ effort\ max\ for\ model\ *\;\ catalog\ does\ not\ advertise\ it) + printf '%s\n' "$line" >&2 + ;; + esac + done <<< "$output" +} diff --git a/bin/fm-remote-secondmate-control.sh b/bin/fm-remote-secondmate-control.sh index e440001aa38..d4fcf343179 100755 --- a/bin/fm-remote-secondmate-control.sh +++ b/bin/fm-remote-secondmate-control.sh @@ -61,6 +61,8 @@ REMOTE_HERDR_SESSION=fm-remote . "$SCRIPT_DIR/fm-backend.sh" # shellcheck source=bin/fm-ff-lib.sh . "$SCRIPT_DIR/fm-ff-lib.sh" +# shellcheck source=bin/fm-codex-catalog-lib.sh +. "$SCRIPT_DIR/fm-codex-catalog-lib.sh" # shellcheck source=bin/fm-pending-reply-lib.sh . "$SCRIPT_DIR/fm-pending-reply-lib.sh" # shellcheck source=bin/fm-task-inbox-lib.sh @@ -203,6 +205,7 @@ cmd_launch() { [ -z "$out" ] || printf '%s\n' "$out" >&2 die "remote host-local secondmate launch failed" fi + fm_codex_catalog_relay_dropped_effort_warnings "$out" [ -f "$meta" ] || die "remote launch returned without endpoint metadata" herdr_session=$(fm_meta_get "$meta" herdr_session) [ "$herdr_session" = "$REMOTE_HERDR_SESSION" ] \ diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index c6086b5c249..d3b485c928e 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -965,6 +965,7 @@ spawn_remote_secondmate() { fi return "$rc" fi + fm_codex_catalog_relay_dropped_effort_warnings "$out" remote_backend=$(printf '%s\n' "$out" | sed -n 's/^backend=//p' | tail -1) remote_target=$(printf '%s\n' "$out" | sed -n 's/^target=//p' | tail -1) remote_harness=$(printf '%s\n' "$out" | sed -n 's/^harness=//p' | tail -1) diff --git a/tests/fm-secondmate-liveness.test.sh b/tests/fm-secondmate-liveness.test.sh index 5d4506cf795..7354cd67f2f 100755 --- a/tests/fm-secondmate-liveness.test.sh +++ b/tests/fm-secondmate-liveness.test.sh @@ -378,6 +378,26 @@ test_sweep_respawns_confirmed_dead_secondmate() { pass "sweep: a confirmed-dead secondmate endpoint is killed and respawned" } +test_sweep_relays_codex_max_downgrade_warning() { + local w fb tmuxfb log out warning count + w=$(new_world sweep-codex-max-warning) + printf '%s\n' 'codex gpt-6-astra max' > "$w/home/config/secondmate-harness" + add_sm_home "$w" sm1 firstmate:fm-sm1 codex + fb=$(make_toolchain "$w"); tmuxfb=$(make_liveness_tmux "$w") + log="$w/calls.log"; : > "$log" + warning='warning: dropped codex effort max for model gpt-6-astra; catalog does not advertise it' + + out=$(run_bootstrap "$tmuxfb:$fb" "$w/home" missing "$log" CODEX_HOME="$w/missing-codex-home") + + assert_contains "$out" "$warning" \ + "a successful recovery should expose the Codex max downgrade warning" + count=$(printf '%s\n' "$out" | grep -Fxc "$warning") + [ "$count" -eq 1 ] || fail "the Codex max downgrade warning should appear once, got $count" + assert_contains "$(cat "$log")" "new-window" \ + "the warning relay must not prevent the secondmate recovery" + pass "sweep: a successful Codex max downgrade warns once" +} + test_sweep_leaves_alive_secondmate_untouched() { local w fb tmuxfb log out w=$(new_world sweep-alive) @@ -553,6 +573,7 @@ test_tmux_agent_state_rejects_malformed_targets_before_probe test_herdr_agent_state_preserves_husk_classifier test_agent_state_dispatcher_and_compatibility test_sweep_respawns_confirmed_dead_secondmate +test_sweep_relays_codex_max_downgrade_warning test_sweep_leaves_alive_secondmate_untouched test_sweep_respawns_authoritatively_missing_pi_secondmate test_sweep_respawns_authoritatively_missing_pi_signed_secondmate diff --git a/tests/fm-secondmate-sync.test.sh b/tests/fm-secondmate-sync.test.sh index 68ca9d80015..1561c7c7ee6 100755 --- a/tests/fm-secondmate-sync.test.sh +++ b/tests/fm-secondmate-sync.test.sh @@ -1342,6 +1342,37 @@ test_remote_launch_does_not_retarget_host_copy() { pass "R9 a remote launch leaves the home on the parent's commit while an ordinary spawn follows its own checkout" } +test_remote_launch_relays_codex_max_downgrade_warning() { + local w c1 herdrbin fakebin launch_out warning count + w=$(new_remote_world remote-launch-codex-max-warning) + c1=$(head_of "$w/main") + add_remote_home "$w" launched "$w/forge.git" "$c1" + mkdir -p "$w/launched/data/.parent-route/launched" + printf '%s\n' '# brief' > "$w/launched/data/.parent-route/launched/brief.md" + + fakebin=$(fm_fakebin "$w/launchfake") + herdrbin="$w/herdrhost" + mkdir -p "$herdrbin/bin" + install_remote_herdr_fixture "$herdrbin" "$w/herdr.state" "$w/herdr.log" \ + "$w/herdr.sendfail" "$w/herdr.sock" + cp "$herdrbin/bin/herdr" "$fakebin/herdr" + fm_fake_exit0 "$fakebin" gh treehouse tmux node + warning='warning: dropped codex effort max for model gpt-6-astra; catalog does not advertise it' + + launch_out=$(PATH="$fakebin:$BASE_PATH" CODEX_HOME="$w/missing-codex-home" \ + FM_HOME="$w/launched" FM_ROOT_OVERRIDE="$w/coderoot" FM_SPAWN_NO_GUARD=1 \ + "$ROOT/bin/fm-remote-secondmate-control.sh" launch launched codex gpt-6-astra max herdr 2>&1) \ + || fail "remote Codex max launch failed: $launch_out" + + assert_contains "$launch_out" "$warning" \ + "a successful remote launch should expose the Codex max downgrade warning" + count=$(printf '%s\n' "$launch_out" | grep -Fxc "$warning") + [ "$count" -eq 1 ] || fail "the remote Codex max downgrade warning should appear once, got $count" + assert_contains "$launch_out" "schema=fm-remote-secondmate-control.v1" \ + "the warning relay must preserve the remote launch route" + pass "remote launch: a successful Codex max downgrade warns once" +} + test_ff_updated test_ff_current test_ff_dirty @@ -1374,5 +1405,6 @@ test_remote_sync_without_target_follows_host_copy test_bootstrap_syncs_remote_home_to_primary_commit test_bootstrap_reports_outdated_host_actionably test_remote_launch_does_not_retarget_host_copy +test_remote_launch_relays_codex_max_downgrade_warning echo "# all fm-secondmate-sync tests passed" From 112c1212cf9897410c522d2cee41f85c57caf64a Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:48:41 +0800 Subject: [PATCH 15/21] no-mistakes(document): Document catalog validation and feedback metadata --- .agents/skills/process-event-sources/SKILL.md | 2 +- bin/fm-procevent-lavish.sh | 8 ++++---- docs/calm-mode-feasibility.md | 4 ++-- docs/configuration.md | 4 +++- docs/verification/process-event-sources.md | 2 +- 5 files changed, 11 insertions(+), 9 deletions(-) diff --git a/.agents/skills/process-event-sources/SKILL.md b/.agents/skills/process-event-sources/SKILL.md index 8a72732e243..0c8b0ec69f8 100644 --- a/.agents/skills/process-event-sources/SKILL.md +++ b/.agents/skills/process-event-sources/SKILL.md @@ -113,7 +113,7 @@ Two rules the commands cannot enforce for you: This call is atomically deduplicated by the exact source and sequence: it prints `handled: ` only the first time and `already-handled: ` on every repeat, so a paired effect gated on that distinction is never authorized twice. Reading the event line or the result file is not handling - only this call durably retires the wake, so call it every time, including on a repeat wake for a sequence you already acted on. : Ask the adapter what the result means rather than parsing it yourself. `bin/fm-procevent.sh classify ` routes through the immutable built-in or extension identity captured with that result; for Lavish, its existing direct command returns `feedback`, `ended`, `waiting`, `disconnected`, `missing`, or `unknown`. - Consume a Lavish capture with `bin/fm-procevent-lavish.sh read ` rather than grepping the raw file: that command reports declared and presented item counts plus a completeness verdict, enumerates every captured queued item while retaining supplied element identity, and surfaces a `tag=message` freeform message as its own field, labeling it as session-ending only when the session ended. + Consume a Lavish capture with `bin/fm-procevent-lavish.sh read ` rather than grepping the raw file: that command reports declared and presented item counts plus a completeness verdict, enumerates every captured queued item while retaining supplied element, target, and attachment metadata, and surfaces a `tag=message` freeform message as its own field, labeling it as session-ending only when the session ended. A count mismatch or malformed item makes `read` report an incomplete result and exit nonzero, so leave that capture unacknowledged. `answers` remains the keyed-choice extractor and never treats freeform prose as a decision key. A `feedback` result can still be the last one a review ever produces, so never assume another wake is coming just because the state is not `ended`. diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 9c9a3403b96..7b99260e23a 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -22,10 +22,10 @@ # own labeled field, printed first and distinct from per-element # annotations; it is labeled SESSION-ENDING MESSAGE only when the # session ended. It accepts both field-declared CSV tables and -# YAML-like item lists, including list items with nested attachment -# metadata. Declared and presented item counts, plus a completeness -# verdict, follow before all annotations so a partial read is -# obvious. A count mismatch or malformed row prints `complete: no` +# YAML-like item lists, including list items with nested target and +# attachment metadata. Declared and presented item counts, plus a +# completeness verdict, follow before all annotations so a partial +# read is obvious. A count mismatch or malformed row prints `complete: no` # and exits nonzero. Each annotation retains its element uid, # selector, tag, and text. A non-choice freeform comment (`prompt`) # is printed as its own field even when a selector is also present diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index 68dab0cdc50..812e1482ea6 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -207,7 +207,7 @@ Every tool registered or supplied by Firstmate under `.pi/extensions` has this d | --- | --- | --- | | `read`, `bash`, `edit`, `write`, `grep`, `find`, `ls` | Calm wrappers for Pi's seven main-session built-ins | Their call and text-result shells hide while Calm is active; ordinary and stock export rendering delegate to Pi's original renderers. | | `fm_watch_arm_pi` | Main-session custom tool in `fm-primary-pi-watch.ts` | Its complete self-rendered shell hides while Calm is active and returns unchanged when Calm is off or stock export rendering is active. | -| `fm_branch_outcomes` | Main-session custom tool in `fm-branch-supervision.ts` | Its complete self-rendered shell hides while Calm is active; when visible, the self-renderer reconstructs Pi's ordinary boxed fallback shell and probes Pi's rendered stock fallback to preserve that installed surface's collapsed or all-line output policy plus expanded state, while stock export rendering deliberately falls through to Pi's structured fallback. | +| `fm_branch_outcomes` | Main-session custom tool in `fm-branch-supervision.ts` | Its complete self-rendered shell hides while Calm is active; when visible, the self-renderer reconstructs Pi's ordinary boxed fallback shell and probes Pi's rendered stock fallback to preserve the installed version's argument display, result-preview policy, and expanded state, while stock export rendering deliberately falls through to Pi's structured fallback. | | `fm_branch_processed` | Main-session custom tool in `fm-branch-supervision.ts` | Its complete self-rendered shell hides while Calm is active, exactly like `fm_branch_outcomes`; when visible, the self-renderer reconstructs Pi's ordinary boxed fallback shell around the one-line acknowledgement result, while stock export rendering deliberately falls through to Pi's structured fallback. | | `fm_branch_report` | Branch-session custom tool supplied directly to `createAgentSession` | It runs only in the headless supervision session and has no main-session `ToolExecutionComponent`; successful execution writes the outcome store and delivers a routine note or exact captain entry through the separately audited delivery path, so the tool cannot emit a dump-shaped row in the captain's transcript. | | branch-local `read` built-in | Branch-session built-in enabled through `createAgentSession` | It runs only in the headless supervision session and has no main-session `ToolExecutionComponent`, so its file output cannot emit a row in the captain's transcript. | @@ -281,7 +281,7 @@ Pi's Calm implementation changed only to consume the shared sprite core, while t ## Regression coverage -`tests/fm-calm-pi-extension.test.sh` compares wrapped and stock renderers and verifies all seven built-ins plus `fm_watch_arm_pi`; `tests/fm-pi-branch-extension.test.sh` verifies `fm_branch_outcomes` Calm toggling, capability-probed all-line versus collapsed stock output, exact expanded output, and export rendering. +`tests/fm-calm-pi-extension.test.sh` compares wrapped and stock renderers and verifies all seven built-ins plus `fm_watch_arm_pi`; `tests/fm-pi-branch-extension.test.sh` verifies `fm_branch_outcomes` Calm toggling, capability-probed argument and result-preview rendering, exact expanded output, and export rendering. Together they exercise redraw of already-rendered tool, thinking, current operational-user, and legacy synthetic rows, and cover every policy class. It covers persisted preference restoration across every session-start reason and a real restart, proves the working-ship presentation and Calm-off stock `Working...` row through a delayed deterministic provider, asserts no Calm status row, verifies operational messages remain exact ordinary user-role session entries and complete exports, and drives genuine 100 by 44, 160 by 36, and 180 by 44 terminal fixtures. A native deterministic `/skill:ahoy` turn produces thinking, tool-call, and tool-result blocks, asserts that the collapsed skill-to-final gap equals the two-row visible-only baseline, expands and re-collapses original thinking, restores Calm-off rendering, verifies persisted hidden history, and repeats the geometry assertion after restart with `terminal.clearOnShrink` explicitly off. diff --git a/docs/configuration.md b/docs/configuration.md index f08bcae0235..82524aaf314 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -548,7 +548,9 @@ The resolver returns an actionable configuration error before any request when s A profile `floor` contains only `scope` and `min_percent`, always uses that profile's provider and matched account, and makes that one candidate ineligible below `min_percent` on the named scope. An absent or unknown named row also makes the candidate unrankable and is reported as an unverifiable floor, not as a known shortfall. `ultra` is native-only: the model-aware validation contract and launch mapping are owned by `bin/fm-harness.sh validate-native-effort` and `bin/fm-spawn.sh` respectively. -For Codex `max`, the launch path passes the setting only when the selected model's entry in `${CODEX_HOME:-~/.codex}/models_cache.json` advertises that reasoning level; otherwise it warns and omits the setting. +Bootstrap and typed dispatch validation accept Codex `max` only when the selected model's entry in `${CODEX_HOME:-~/.codex}/models_cache.json` advertises that reasoning level. +The launch path applies the same check and passes the setting when supported. +For direct and recovery launches, it warns once and omits the setting when support is absent. An omitted model or effort means the selected harness uses its own default for that axis. Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. diff --git a/docs/verification/process-event-sources.md b/docs/verification/process-event-sources.md index 4054855c2ec..12f3cdfa129 100644 --- a/docs/verification/process-event-sources.md +++ b/docs/verification/process-event-sources.md @@ -104,7 +104,7 @@ Exercised by `tests/fm-procevent.test.sh` against a fake blocking source whose c | adapter-owned silence verdict | an ordinary firstmate-owned Lavish source driven against a stand-in poll that returns an empty ended session captures its result, records it durably handled, appends no wake, and stays silent through a later `reconcile` that would otherwise republish it, while still retiring its ended source; the same real path with a `Send & End` response carrying the captain's choice still publishes its `check` wake and is left unacknowledged for the handler | | worker-owned Lavish rounds | one three-round fixture arms a board for an identity-matched task endpoint, delivers nonterminal and terminal captures directly to that task's steering inbox without a firstmate `check` wake, acknowledges each nonterminal round through a successful re-arm, redelivers an inbox note filed before acknowledgement, refuses a second armer and every early retirement, and concludes the terminal round through `handled` without another poll; focused fixtures also pin failed re-arm rollback, generation-specific reply staging, one reply post across transient poll retries, unreachable-owner refusal, interrupted conclusion recovery, and repeat acknowledgement isolation | | Lavish handled-status classification | an executable fixture table pins exact `feedback`, `ended`, `waiting`, and `browser_disconnected` mappings, including `browser_disconnected` to `disconnected`; the same suite proves that status is nonterminal and receives a zero-answer silence verdict | -| Lavish result decoding | `tests/fm-procevent.test.sh` and `tests/fm-captain-hold-lifecycle.test.sh` use synthetic field-declared tables and YAML-like item lists to prove that `read` preserves every parsed row, `answers` and `reconciles` extract their respective choice rows from either form, and `silent` recognizes either representation as content; list coverage includes annotations, versioned choice context, nested attachment metadata, and a session-ending message; both representations preserve non-ASCII comments, answers, and reconcile notes; declared-count mismatches preserve parsed rows, report incomplete, and make `read` exit nonzero; a malformed row in either representation also reports incomplete without hiding valid neighboring fields or rows | +| Lavish result decoding | `tests/fm-procevent.test.sh` and `tests/fm-captain-hold-lifecycle.test.sh` use synthetic field-declared tables and YAML-like item lists to prove that `read` preserves every parsed row, `answers` and `reconciles` extract their respective choice rows from either form, and `silent` recognizes either representation as content; list coverage includes annotations, versioned choice context, nested target and attachment metadata, and a session-ending message; both representations preserve non-ASCII comments, answers, and reconcile notes; declared-count mismatches preserve parsed rows, report incomplete, and make `read` exit nonzero; a malformed row in either representation also reports incomplete without hiding valid neighboring fields or rows | | session-derived Lavish routing | the three-round worker fixture starts its first listener under conflicting ambient host/port values and configuration, then recovers later listeners while that conflicting configuration remains, and proves every reply/poll uses the board's saved session endpoint; direct polls cover Unicode artifact paths, hostnames, IPv6, session endpoint changes, quiet retries, and refusal before reply consumption when session evidence is absent or invalid; spawn coverage still proves the configured opening address enters the worker launch | | silence fails closed | the adapter's published `silent` command suppresses only an `ended` session with no queued content block or a `browser_disconnected` response, and announces a real answer, freeform prose, any recognized content block regardless of its declared count, a malformed top-level content header, a `waiting` or `missing` session, a server error, an unreadable result, and indented payload text imitating an empty content block; the `remote-reply` and `when` adapters, which implement no `silent` command, announce every result | | terminal retirement preserves the result | the retired source's captured output, its announced event, its handled acknowledgement, and later explicit `retire` all still behave normally | From ca58502396254d28bc473fe66eaa747c2ae340c4 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 08:58:00 +0800 Subject: [PATCH 16/21] no-mistakes(review): Relay Codex max warnings through secondmate restarts --- bin/fm-secondmate-restart.sh | 3 +++ tests/fm-secondmate-restart.test.sh | 34 +++++++++++++++++++++++++++++ 2 files changed, 37 insertions(+) diff --git a/bin/fm-secondmate-restart.sh b/bin/fm-secondmate-restart.sh index 7eccea3b0e2..7704a6eb091 100755 --- a/bin/fm-secondmate-restart.sh +++ b/bin/fm-secondmate-restart.sh @@ -87,6 +87,8 @@ STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" # shellcheck source=bin/fm-secondmate-restart-lib.sh . "$SCRIPT_DIR/fm-secondmate-restart-lib.sh" +# shellcheck source=bin/fm-codex-catalog-lib.sh +. "$SCRIPT_DIR/fm-codex-catalog-lib.sh" # shellcheck source=bin/fm-secondmate-nudge-lib.sh . "$SCRIPT_DIR/fm-secondmate-nudge-lib.sh" # shellcheck source=bin/fm-pending-reply-lib.sh @@ -209,6 +211,7 @@ restart_mate() { # restart_rc=$? fi if [ "$restart_rc" -eq 0 ]; then + fm_codex_catalog_relay_dropped_effort_warnings "$restart_out" ran_on=$(printf '%s\n' "$restart_out" | sed -n 's/^relaunched .* harness=\([^ ]*\).*/\1/p' | tail -1) [ -n "$ran_on" ] || ran_on=${HARNESS[i]} if [ "${PLACEMENT[i]}" = remote ]; then diff --git a/tests/fm-secondmate-restart.test.sh b/tests/fm-secondmate-restart.test.sh index 2eb2b5857f0..df4e3e6220a 100755 --- a/tests/fm-secondmate-restart.test.sh +++ b/tests/fm-secondmate-restart.test.sh @@ -521,6 +521,9 @@ case "${rargs[1]:-}" in /bin/sleep 2 : > "$FM_FAKE_DIR/remote-relaunch-end" ;; + warning) + printf 'warning: dropped codex effort max for model gpt-6-astra; catalog does not advertise it\n' + ;; esac printf 'relaunched %s\n' "${rargs[2]}" ;; @@ -594,6 +597,36 @@ test_local_restart_uses_the_home_pin_and_reports_what_ran() { pass "T8 a local restart re-resolves this home's pin and reports the runtime that came up" } +test_codex_max_warning_survives_local_and_remote_restarts() { + local dir out rc warning count + warning='warning: dropped codex effort max for model gpt-6-astra; catalog does not advertise it' + + dir=$(new_case codex-warning-local) + add_local_mate "$dir" sm1 + arm_answer "$dir" sm1 + printf 'codex gpt-6-astra max\n' > "$dir/home/config/secondmate-harness" + printf 'codex' > "$dir/fake/becomes" + out=$(CODEX_HOME="$dir/missing-codex-home" run_restart "$dir" sm1); rc=$? + + expect_code 0 "$rc" "a local Codex restart with a missing catalog should succeed: $out" + assert_contains "$out" "$warning" "the local restart swallowed the Codex max downgrade warning" + count=$(printf '%s\n' "$out" | grep -Fxc "$warning") + [ "$count" -eq 1 ] || fail "the local restart should relay one Codex max downgrade warning, got $count" + + dir=$(new_case codex-warning-remote) + setup_remote_case "$dir" sm2 warning + export FM_FAKE_ANSWER_STATUS="$dir/home/state/sm2.status" + printf 'codex gpt-6-astra max\n' > "$dir/home/config/secondmate-harness" + out=$(run_restart "$dir" sm2); rc=$? + unset FM_FAKE_ANSWER_STATUS + + expect_code 0 "$rc" "a remote Codex restart with a missing catalog should succeed: $out" + assert_contains "$out" "$warning" "the remote restart swallowed the Codex max downgrade warning" + count=$(printf '%s\n' "$out" | grep -Fxc "$warning") + [ "$count" -eq 1 ] || fail "the remote restart should relay one Codex max downgrade warning, got $count" + pass "Codex max downgrade warnings survive local and remote restart capture" +} + test_native_ultra_restart_keeps_local_and_remote_profiles() { local dir out rc relaunch_line dir=$(new_case native-local) @@ -982,6 +1015,7 @@ test_unprovable_runtime_falls_back test_unknown_mate_is_accounted_for test_refused_restart_falls_back_without_claiming_a_reload test_local_restart_uses_the_home_pin_and_reports_what_ran +test_codex_max_warning_survives_local_and_remote_restarts test_native_ultra_restart_keeps_local_and_remote_profiles test_remote_mate_restarts_over_the_transport_hop test_unreachable_host_is_reported_unknown From fce2fe2e15267d56a4ec2b13f5e973633e8b446d Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:10:51 +0800 Subject: [PATCH 17/21] no-mistakes(review): Reject truncated Lavish list items as incomplete --- bin/fm-procevent-lavish.sh | 14 +++++++++++--- tests/fm-procevent.test.sh | 15 +++++++++++++++ 2 files changed, 26 insertions(+), 3 deletions(-) diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 7b99260e23a..18850bcba4e 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -561,8 +561,15 @@ prompt_rows_json() { # } return @values; } - my ($current, $line, $target_lines, $attachment_rows); + my ($current, $line, $target_lines, $attachment_rows, $malformed_at_start); my @attachment_fields; + my $finish_current = sub { + return unless defined $current; + my $missing = grep { !exists $current->{$_} } qw(uid prompt selector tag text); + $malformed++ if $missing && $malformed == $malformed_at_start; + push @rows, $current; + undef $current; + }; while (defined($line = <$fh>)) { if (!$header) { if ($line =~ /^(?:prompts|feedback)\[(\d+)\]\{([^}]*)\}:\s*$/) { @@ -609,8 +616,9 @@ prompt_rows_json() { # undef $target_lines; } if ($line =~ /^ -\s+(.+)$/) { - push @rows, $current if defined $current; + $finish_current->(); $current = {}; + $malformed_at_start = $malformed; if ($1 =~ /^([A-Za-z_][A-Za-z0-9_]*):\s*(.*)$/) { $current->{$1} = unquote($2); } else { $malformed++; } @@ -660,7 +668,7 @@ prompt_rows_json() { # } $malformed++ if defined($attachment_rows) && $attachment_rows; $malformed++ if defined($target_lines) && !@$target_lines; - push @rows, $current if defined $current; + $finish_current->(); close $fh; print encode_json({declared => $header ? 0 + $declared : 0, rows => \@rows, malformed => 0 + $malformed, header => 0 + $header}); diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index b83e6ee8469..4c8fc20d221 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -3284,6 +3284,21 @@ assert_contains "$out" "complete: no" \ "a malformed list-form field was certified as complete" pass "read never certifies malformed list-form fields as complete" +cat > "$READ" <<'EOF' +session: + status: feedback +prompts[1]: + - uid: "el-a" +EOF +read_status=0 +out=$(read_out 2>&1) || read_status=$? +[ "$read_status" -ne 0 ] || fail "read certified a truncated list item as complete" +assert_contains "$out" "declared_items: 1" "a truncated list item lost its declared count" +assert_contains "$out" "presented_items: 1" "a truncated list item was not presented" +assert_contains "$out" "malformed_items: 1" "a truncated list item was not marked malformed" +assert_contains "$out" "complete: no" "a truncated list item was certified as complete" +pass "read rejects list items missing required fields" + cat > "$READ" <<'EOF' session: file: /review.html From 5bf3d8794ad9dc23b3ec2a47dc3cdf19689c3375 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:45:31 +0800 Subject: [PATCH 18/21] no-mistakes(document): Document Codex catalog helper --- docs/scripts.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/scripts.md b/docs/scripts.md index 48bdce2d641..685e2d692b0 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -60,6 +60,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | [`fm-project-origin-lib.sh`](../bin/fm-project-origin-lib.sh) | Accepted origin-form owner shared by both remote provisioning boundaries | | `fm-landing-remote.sh` | Point `origin` at the repository work lands on and name the third-party parent `upstream` | | `fm-spawn.sh` | Spawn crewmates, scouts, `id=repo` batches, and secondmates on the resolved harness and runtime backend | +| `fm-codex-catalog-lib.sh` | Read installed Codex model effort support and relay warnings when unsupported `max` requests are omitted | | `fm-backend.sh` | Runtime-backend selection, meta helpers, selector resolution, and operation dispatch | | `fm-backend-hometag-lib.sh` | Shared per-installation home-tag derivation for zellij tab and cmux workspace titles | | `fm-composer-lib.sh` | Single fleet-wide owner of composer shapes, capability-aware screen classification, and verdicts | From 68b48e7a8896b5f25b051af32c383aa28a054ffa Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 10:03:20 +0800 Subject: [PATCH 19/21] no-mistakes(review): Validate Codex catalog schema before enabling max --- bin/fm-codex-catalog-lib.sh | 41 +++++++++++++++++++------ tests/fm-dispatch-resolve.test.sh | 21 +++++++++++++ tests/fm-spawn-dispatch-profile.test.sh | 21 +++++++++++++ 3 files changed, 73 insertions(+), 10 deletions(-) diff --git a/bin/fm-codex-catalog-lib.sh b/bin/fm-codex-catalog-lib.sh index 9b017290870..dfaf22c604a 100755 --- a/bin/fm-codex-catalog-lib.sh +++ b/bin/fm-codex-catalog-lib.sh @@ -4,29 +4,50 @@ fm_codex_catalog_path() { printf '%s\n' "${CODEX_HOME:-$HOME/.codex}/models_cache.json" } +fm_codex_catalog_read() { + local catalog + command -v jq >/dev/null 2>&1 || return 1 + catalog=$(fm_codex_catalog_path) + jq -ces ' + def valid_reasoning_level: + if type != "object" then false + else (.effort | type) == "string" + end; + def valid_model: + if type != "object" then false + elif (.supported_reasoning_levels | type) != "array" then false + else all(.supported_reasoning_levels[]; valid_reasoning_level) + end; + if length != 1 then error("expected one catalog object") + elif (.[0] | type) != "object" then error("catalog must be an object") + elif (.[0].models | type) != "array" then error("catalog models must be an array") + elif (.[0].models | all(.[]; valid_model) | not) then error("invalid catalog model") + else .[0] + end + ' "$catalog" 2>/dev/null +} + fm_codex_catalog_supports_effort() { local model=$1 effort=$2 catalog [ -n "$model" ] && [ "$model" != default ] || return 1 - command -v jq >/dev/null 2>&1 || return 1 - catalog=$(fm_codex_catalog_path) + catalog=$(fm_codex_catalog_read) || return 1 jq -e --arg model "$model" --arg effort "$effort" ' - any(.models[]?; + any(.models[]; (.slug? == $model) - and any(.supported_reasoning_levels[]?; .effort? == $effort) + and any(.supported_reasoning_levels[]; .effort == $effort) ) - ' "$catalog" >/dev/null 2>&1 + ' <<< "$catalog" >/dev/null 2>&1 } fm_codex_catalog_models_supporting_effort() { local effort=$1 catalog models - command -v jq >/dev/null 2>&1 || return 1 - catalog=$(fm_codex_catalog_path) + catalog=$(fm_codex_catalog_read) || return 1 models=$(jq -r --arg effort "$effort" ' - .models[]? - | select(any(.supported_reasoning_levels[]?; .effort? == $effort)) + .models[] + | select(any(.supported_reasoning_levels[]; .effort == $effort)) | .slug? | select(type == "string" and length > 0) - ' "$catalog" 2>/dev/null) || return 1 + ' <<< "$catalog" 2>/dev/null) || return 1 [ -z "$models" ] || printf '%s\n' "$models" } diff --git a/tests/fm-dispatch-resolve.test.sh b/tests/fm-dispatch-resolve.test.sh index 41f10c77374..3d3e2ff335c 100755 --- a/tests/fm-dispatch-resolve.test.sh +++ b/tests/fm-dispatch-resolve.test.sh @@ -741,6 +741,27 @@ assert_contains "$err" 'each use profile effort must be supported by its harness "partial catalog output authorized Codex max" assert_absent "$LOG/argv" "a malformed catalog reached dispatch resolution" +printf '%s\n' \ + '{"models":{"entry":{"slug":"gpt-6-astra","supported_reasoning_levels":{"level":{"effort":"max"}}}}}' \ + > "$CODEX_CATALOG/models_cache.json" +reset_log +CODEX_HOME="$CODEX_CATALOG" TYPESAFE_API_KEY=$KEY run code out err "$BRIEF" +expect_code 2 "$code" "an object-shaped catalog rejects Codex max" +assert_contains "$err" 'each use profile effort must be supported by its harness and model' \ + "object-shaped catalog containers authorized Codex max" +assert_absent "$LOG/argv" "an object-shaped catalog reached dispatch resolution" + +printf '%s\n' \ + '{"models":[{"slug":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ + '{"models":[{"slug":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ + > "$CODEX_CATALOG/models_cache.json" +reset_log +CODEX_HOME="$CODEX_CATALOG" TYPESAFE_API_KEY=$KEY run code out err "$BRIEF" +expect_code 2 "$code" "multiple catalog documents reject Codex max" +assert_contains "$err" 'each use profile effort must be supported by its harness and model' \ + "multiple catalog documents authorized Codex max" +assert_absent "$LOG/argv" "multiple catalog documents reached dispatch resolution" + printf '%s\n' \ '{"models":[{"slug":"other-model","id":"gpt-6-astra","model":"gpt-6-astra","supported_reasoning_levels":[{"effort":"max"}]}]}' \ > "$CODEX_CATALOG/models_cache.json" diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 2edeadad8ee..d30d466014e 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -468,6 +468,26 @@ test_codex_omits_max_effort_for_unsupported_model() { pass "codex omits max for models without the catalog capability" } +test_codex_warns_and_omits_max_for_malformed_catalog() { + local rec id out status launch warning + id=profile-codex-max-malformed-z4c + rec=$(make_spawn_case profile-codex-max-malformed codex "$id") + read_case_record "$rec" + seed_codex_catalog "$HOME_DIR" \ + '{"models":{"entry":{"slug":"gpt-6-astra","supported_reasoning_levels":{"level":{"effort":"max"}}}}}' + warning='warning: dropped codex effort max for model gpt-6-astra; catalog does not advertise it' + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR" --model gpt-6-astra --effort max 2>&1) + status=$? + expect_code 0 "$status" "codex spawn with a malformed catalog should omit max effort" + assert_contains "$out" "$warning" "codex spawn silently downgraded max for a malformed catalog" + launch=$(cat "$LAUNCH_LOG") + assert_contains "$launch" "codex --model 'gpt-6-astra' --dangerously-bypass-approvals-and-sandbox" \ + "codex launch did not preserve the model when a malformed catalog rejected max" + assert_not_contains "$launch" "model_reasoning_effort" "a malformed catalog authorized Codex max" + pass "codex warns and omits max for malformed catalogs" +} + # Codex parks a crewmate launch forever on its unanswerable hook-trust modal # unless the launch turns the hook layer off. These two cases pin the split: # a crewmate runs hook-free, a secondmate keeps the project hooks that carry its @@ -1510,6 +1530,7 @@ test_claude_threads_model_and_effort test_codex_threads_model_and_effort test_codex_threads_model_and_max_effort test_codex_omits_max_effort_for_unsupported_model +test_codex_warns_and_omits_max_for_malformed_catalog test_codex_crewmate_launch_disables_the_hook_layer test_codex_secondmate_launch_keeps_the_hook_layer test_grok_threads_model_and_reasoning_effort From d393b4c221b9f0cab1f115884d585cc4fd0d5b57 Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 10:31:40 +0800 Subject: [PATCH 20/21] no-mistakes(document): Document Codex catalog validation and fallback --- .agents/skills/harness-adapters/references/harness/codex.md | 2 +- docs/configuration.md | 4 +++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/.agents/skills/harness-adapters/references/harness/codex.md b/.agents/skills/harness-adapters/references/harness/codex.md index cba029962e7..dfef29e118d 100644 --- a/.agents/skills/harness-adapters/references/harness/codex.md +++ b/.agents/skills/harness-adapters/references/harness/codex.md @@ -12,7 +12,7 @@ Verified on 2026-06-11 with codex-cli 0.139.0 unless a fact gives a newer versio | Skill invocation | `$`, for example `$no-mistakes`; `/` is Claude-only and Codex rejects it as "Unrecognized command". | | Resume | `codex resume `, using the id printed on quit. | | Model flag | `--model `. | -| Effort flag | `-c 'model_reasoning_effort=""'`; for `max`, Firstmate passes the setting only when the selected model's entry in `${CODEX_HOME:-~/.codex}/models_cache.json` advertises `max`, otherwise it warns and omits it. | +| Effort flag | `-c 'model_reasoning_effort=""'`; the [configuration guide](../../../../../docs/configuration.md#crew-dispatch-profiles-configcrew-dispatchjson) owns Firstmate's catalog-gated `max` validation and launch fallback. | | Model discovery | Open the current interactive session's `/model` picker. | | Marker | None; identity comes from ancestry, and `../../../bin/fm-harness.sh` is what keeps a retained foreign `CLAUDECODE` from renaming it. Verified on 2026-09-01 with codex-cli 0.152.0: the pane process is the `node` npm shim and the native `codex` binary runs as its foreground child, so a tool subprocess reaches the native name directly while the shim itself is identified from its script path. | diff --git a/docs/configuration.md b/docs/configuration.md index 82524aaf314..fcaf3f3a126 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -550,7 +550,9 @@ An absent or unknown named row also makes the candidate unrankable and is report `ultra` is native-only: the model-aware validation contract and launch mapping are owned by `bin/fm-harness.sh validate-native-effort` and `bin/fm-spawn.sh` respectively. Bootstrap and typed dispatch validation accept Codex `max` only when the selected model's entry in `${CODEX_HOME:-~/.codex}/models_cache.json` advertises that reasoning level. The launch path applies the same check and passes the setting when supported. -For direct and recovery launches, it warns once and omits the setting when support is absent. +A missing, unreadable, or malformed catalog cannot authorize `max`, including a catalog that is not one JSON object with a `models` array whose model entries each contain a `supported_reasoning_levels` array. +Bootstrap and typed dispatch validation reject the profile in those cases. +Direct and recovery launches warn once and omit the setting when the catalog cannot authorize it. An omitted model or effort means the selected harness uses its own default for that axis. Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. From 72dbec0bbadb19d00052ec70f74f4a7bbc9f98fb Mon Sep 17 00:00:00 2001 From: PP <121104417+BohnBawerick@users.noreply.github.com> Date: Sat, 3 Oct 2026 11:37:02 +0800 Subject: [PATCH 21/21] no-mistakes(ci): Fixed both CI failures. Updated the Lavish text-range assertion to match nested path metadata output and added fm-codex-catalog-lib.sh to the synthetic remote-root fixture. Both affected test files pass locally. Bash syntax and git diff checks also pass --- tests/fm-procevent.test.sh | 2 +- tests/fm-remote-transport-lanes.test.sh | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index 4c8fc20d221..622fd25a83e 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -3248,7 +3248,7 @@ EOF out=$(read_out) || fail "read rejected valid target and attachment metadata" assert_contains "$out" "presented_items: 6" "target-bearing prompts were dropped" assert_contains "$out" "malformed_items: 0" "valid nested target metadata was marked malformed" -assert_contains "$out" "| path[2]: 0,1" "text-range path metadata was dropped" +assert_contains "$out" "| path[2]: 0,1" "text-range path metadata was dropped" assert_contains "$out" "| rowLabel: Pro" "table-cell target metadata was dropped" assert_contains "$out" "| nodeId: queue" "Mermaid target metadata was dropped" assert_contains "$out" "| scenePath: /tmp/review/0.excalidraw" "whiteboard scene path was dropped" diff --git a/tests/fm-remote-transport-lanes.test.sh b/tests/fm-remote-transport-lanes.test.sh index cbdf1356092..3228ca84e34 100755 --- a/tests/fm-remote-transport-lanes.test.sh +++ b/tests/fm-remote-transport-lanes.test.sh @@ -55,7 +55,8 @@ cp "$ROOT/bin/fm-remote-job-lib.sh" "$ROOT/bin/fm-remote-job-worker.sh" \ "$ROOT/bin/fm-operational-input.sh" "$ROOT/bin/fm-tmux-lib.sh" \ "$ROOT/bin/fm-composer-lib.sh" "$ROOT/bin/fm-cursor-lib.sh" \ "$ROOT/bin/fm-classify-lib.sh" "$ROOT/bin/fm-timeout-lib.sh" \ - "$ROOT/bin/fm-ff-lib.sh" "$ROOT/bin/fm-secondmate-registry-lib.sh" \ + "$ROOT/bin/fm-ff-lib.sh" "$ROOT/bin/fm-codex-catalog-lib.sh" \ + "$ROOT/bin/fm-secondmate-registry-lib.sh" \ "$REMOTE_ROOT/bin/" mkdir -p "$REMOTE_ROOT/bin/backends" cp "$ROOT/bin/backends/herdr.sh" "$REMOTE_ROOT/bin/backends/herdr.sh"