From cf75e96e2f0579b23254bb818429f9cd9d033837 Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Sun, 23 Aug 2026 10:16:19 -0700 Subject: [PATCH 1/2] Add tests for bench-list preview sanitization (CL-6735) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sidebar/bench activity must never preview credential-error paragraphs, HTTP status, or raw provider dumps — prefer last good text, else the short consumer notice. --- packages/chat/src/room-messages.test.ts | 105 +++++++++++++++++++++++- 1 file changed, 101 insertions(+), 4 deletions(-) diff --git a/packages/chat/src/room-messages.test.ts b/packages/chat/src/room-messages.test.ts index 10c62adb4..a3b5da715 100644 --- a/packages/chat/src/room-messages.test.ts +++ b/packages/chat/src/room-messages.test.ts @@ -1,5 +1,6 @@ import { describe, expect, test } from "bun:test"; +import { CONSUMER_INFERENCE_FAILURE_NOTICE } from "./consumer-inference-text"; import { createInMemoryRoomMessageStore, postRoomMessage, @@ -217,6 +218,84 @@ describe("listActivity", () => { // A workbench with no messages is absent, never a fabricated zero. expect(activity["run_never_opened"]).toBeUndefined(); }); + + // CL-6735: bench-list preview never shows the credential-error paragraph + // (or HTTP/raw dumps). Prefer the last good human/agent text; fall back + // to the short consumer notice when nothing earlier qualifies. + // CL-6795 (preview blanks/ignores the latest user message) is separate — + // this only skips failed-turn / classified-failure copy, never a normal + // human or agent reply. + test("a failed turn previews the last good text, never the failure paragraph", async () => { + const roomMessages = createInMemoryRoomMessageStore(); + const publisher = recordingPublisher(); + const post = ( + parts: Parameters[1]["parts"], + sender: { name: null; address: string } = { + name: null, + address: "prn_ada@acme.example", + }, + ) => + postRoomMessage( + { roomMessages, publish: publisher.publish }, + { tenantId: TENANT, workbenchId: WORKBENCH, sender, parts }, + ); + + await post([{ kind: "text", text: "draft the agenda" }]); + await Bun.sleep(2); + const failed = await post( + [ + { + kind: "text", + text: "This agent could not complete your request due to a credential error [HTTP 401]: API key is invalid.", + turnFailed: true, + }, + ], + { name: null, address: "run_myra@acme.example" }, + ); + + const activity = await roomMessages.listActivity({ + tenantId: TENANT, + workbenches: [{ workbenchId: WORKBENCH }], + }); + + expect(activity[WORKBENCH]?.lastActivityAt).toBe(failed.createdAt); + expect(activity[WORKBENCH]?.preview).toBe("draft the agenda"); + expect(activity[WORKBENCH]?.preview).not.toMatch(/credential error/i); + expect(activity[WORKBENCH]?.preview).not.toMatch(/\[HTTP/i); + expect(activity[WORKBENCH]?.preview).not.toContain("API key"); + }); + + test("a lone failed turn falls back to the short consumer notice", async () => { + const roomMessages = createInMemoryRoomMessageStore(); + const publisher = recordingPublisher(); + await postRoomMessage( + { roomMessages, publish: publisher.publish }, + { + tenantId: TENANT, + workbenchId: WORKBENCH, + sender: { name: null, address: "run_myra@acme.example" }, + runId: "run_myra", + parts: [ + { + kind: "text", + text: "I can't reach a model right now — add or check your model key in Settings, then I'll pick this up. (ref abc)", + turnFailed: true, + }, + ], + }, + ); + + const activity = await roomMessages.listActivity({ + tenantId: TENANT, + workbenches: [{ workbenchId: WORKBENCH }], + }); + + expect(activity[WORKBENCH]?.preview).toBe( + CONSUMER_INFERENCE_FAILURE_NOTICE, + ); + expect(activity[WORKBENCH]?.preview).not.toMatch(/model key/i); + expect(activity[WORKBENCH]?.preview).not.toMatch(/ref /i); + }); }); describe("previewOf", () => { @@ -240,7 +319,7 @@ describe("previewOf", () => { ).toBe(""); }); - test("does not preview HTTP status or raw provider dumps", () => { + test("does not preview HTTP status, raw provider dumps, or the failure paragraph", () => { expect( previewOf([ { @@ -248,8 +327,26 @@ describe("previewOf", () => { text: "This agent could not complete your request due to a credential error [HTTP 401]: API key is invalid.", }, ]), - ).toBe( - "This agent could not complete your request due to a credential error", - ); + ).toBe(CONSUMER_INFERENCE_FAILURE_NOTICE); + expect( + previewOf([ + { + kind: "text", + text: "This agent could not complete your request because the API quota has been exhausted [HTTP 429]: rate limited", + }, + ]), + ).toBe(CONSUMER_INFERENCE_FAILURE_NOTICE); + }); + + test("a turnFailed notice is never the bench-list preview copy", () => { + expect( + previewOf([ + { + kind: "text", + text: "I didn't get that one — send it again and I'll pick it up. (ref xyz)", + turnFailed: true, + }, + ]), + ).toBe(CONSUMER_INFERENCE_FAILURE_NOTICE); }); }); From 46ff279357a0f5fa63e511ffc0debaa0bd61de37 Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Sun, 23 Aug 2026 10:16:25 -0700 Subject: [PATCH 2/2] Sanitize bench-list previews away from failure paragraphs (CL-6735) previewOf and listActivity collapse classified inference failures and turnFailed notices to the short consumer sentence, and walk recent messages for the last good human/agent text when the newest is a failure. --- packages/chat/src/consumer-inference-text.ts | 32 ++++++ packages/chat/src/room-messages.ts | 115 ++++++++++++++++--- 2 files changed, 133 insertions(+), 14 deletions(-) diff --git a/packages/chat/src/consumer-inference-text.ts b/packages/chat/src/consumer-inference-text.ts index da999311c..54f4c962e 100644 --- a/packages/chat/src/consumer-inference-text.ts +++ b/packages/chat/src/consumer-inference-text.ts @@ -10,6 +10,18 @@ const TRAILING_HTTP_DUMP = /\s*\[HTTP\s+\d+\]:[\s\S]*$/i; export const CONSUMER_INFERENCE_FAILURE_NOTICE = "This didn't go through. Try again, or check the connection in Settings."; +/** + * Byte-for-byte copies of the two preambles `@intx/inference`'s + * `formatInferenceError` writes for `credential_failure` and + * `quota_exhausted`. Kept here so the bench-list preview path (CL-6735) + * can refuse them without depending on `@corbits/chat-ui`. The chat-ui + * drift guard still owns matching these against the published director. + */ +export const CLASSIFIED_INFERENCE_FAILURE_PREAMBLES: readonly string[] = [ + "This agent could not complete your request due to a credential error", + "This agent could not complete your request because the API quota has been exhausted", +]; + function isProviderJsonDump(raw: string): boolean { const trimmed = raw.trim(); if (!(trimmed.startsWith("{") || trimmed.startsWith("["))) return false; @@ -27,3 +39,23 @@ export function consumerFacingInferenceText(raw: string): string { if (stripped.length > 0 && !needsSanitization(stripped)) return stripped; return CONSUMER_INFERENCE_FAILURE_NOTICE; } + +/** True when `text` is (or starts with) a classified inference-failure preamble. */ +export function isClassifiedInferenceFailureText(text: string): boolean { + return CLASSIFIED_INFERENCE_FAILURE_PREAMBLES.some((preamble) => + text.startsWith(preamble), + ); +} + +/** + * Bench-list / sidebar preview copy (CL-6735): never the full failure + * paragraph, never HTTP/raw provider dumps — a short consumer sentence + * when the text is a classified failure. + */ +export function activityPreviewText(raw: string): string { + const facing = consumerFacingInferenceText(raw); + if (isClassifiedInferenceFailureText(facing)) { + return CONSUMER_INFERENCE_FAILURE_NOTICE; + } + return facing; +} diff --git a/packages/chat/src/room-messages.ts b/packages/chat/src/room-messages.ts index afba9755a..c67efdaa1 100644 --- a/packages/chat/src/room-messages.ts +++ b/packages/chat/src/room-messages.ts @@ -14,7 +14,13 @@ import { and, asc, count, desc, eq, gt, inArray, lt, or } from "drizzle-orm"; import type { Part } from "./parts"; import { workbenchMessages } from "./schema"; import type { ChatDb } from "./store"; -import { consumerFacingInferenceText } from "./consumer-inference-text"; +import { + activityPreviewText, + CONSUMER_INFERENCE_FAILURE_NOTICE, + consumerFacingInferenceText, + isClassifiedInferenceFailureText, +} from "./consumer-inference-text"; + import { ChatMessageEventData } from "./stream-events"; import type { WorkbenchSubscriberRegistry } from "./workbench-events"; @@ -99,6 +105,8 @@ export interface RoomMessageStore { * has always read a workbench in. */ const PAGE_SIZE = 50; +/** Cap on how far back listActivity walks for a last-good preview (CL-6735). */ +const PREVIEW_LOOKBACK = 20; const PREVIEW_MAX_LENGTH = 80; let lastMintedAt = 0; @@ -128,10 +136,16 @@ function newMessageId(): string { /** * A bounded preview of a message for a workbench-list row: its text * parts, whitespace-collapsed and truncated. An attachment-only message - * previews as nothing rather than a fabricated placeholder. + * previews as nothing rather than a fabricated placeholder. Failed turns + * and classified inference-failure paragraphs (CL-6735) collapse to the + * short consumer notice — never HTTP status, raw provider dumps, or the + * full credential-error sentence. */ export function previewOf(parts: readonly Part[]): string { - const text = consumerFacingInferenceText( + if (isFailurePreviewParts(parts)) { + return CONSUMER_INFERENCE_FAILURE_NOTICE; + } + const text = activityPreviewText( parts .filter((part): part is Extract => { return part.kind === "text"; @@ -146,6 +160,55 @@ export function previewOf(parts: readonly Part[]): string { : text; } +/** True when these parts must not appear as a bench-list preview. */ +function isFailurePreviewParts(parts: readonly Part[]): boolean { + if (parts.some((part) => part.kind === "text" && part.turnFailed === true)) { + return true; + } + const joined = parts + .filter((part): part is Extract => { + return part.kind === "text"; + }) + .map((part) => part.text) + .join(" ") + .replace(/\s+/g, " ") + .trim(); + if (joined.length === 0) return false; + return isClassifiedInferenceFailureText(consumerFacingInferenceText(joined)); +} + +/** + * Pick bench-list preview text from newest-first messages: skip failed + * turns and classified failure paragraphs, keep the last good human/agent + * text, and fall back to the short consumer notice when nothing else + * qualifies (CL-6735). Does not skip ordinary user/agent replies + * (CL-6795 is separate). + */ +function activityPreviewFromNewestFirst( + newestFirst: readonly RoomMessage[], +): string { + for (const message of newestFirst) { + if (isFailurePreviewParts(message.parts)) continue; + const preview = previewOf(message.parts); + if (preview.length > 0) return preview; + } + const newest = newestFirst[0]; + if (newest !== undefined && isFailurePreviewParts(newest.parts)) { + return CONSUMER_INFERENCE_FAILURE_NOTICE; + } + return ""; +} + +function summaryOf( + newest: RoomMessage, + unreadCount: number, + newestFirstForPreview: readonly RoomMessage[] = [newest], +): RoomActivitySummary { + const preview = activityPreviewFromNewestFirst(newestFirstForPreview); + const base = { unreadCount, lastActivityAt: newest.createdAt }; + return preview.length === 0 ? base : { ...base, preview }; +} + /** * Puts a message on a workbench's timeline: one durable row, then one * event onto the live stream so every open client sees it without asking @@ -214,15 +277,6 @@ function pageOf(newestFirst: readonly RoomMessage[]): ListedRoomMessages { : { items }; } -function summaryOf( - newest: RoomMessage, - unreadCount: number, -): RoomActivitySummary { - const preview = previewOf(newest.parts); - const base = { unreadCount, lastActivityAt: newest.createdAt }; - return preview.length === 0 ? base : { ...base, preview }; -} - interface MessageRow { id: string; workbenchId: string; @@ -374,9 +428,31 @@ export function createDrizzleRoomMessageStore( const result: Record = {}; for (const row of newestRows) { const newest = toRoomMessage(row as MessageRow); + const unreadCount = unreadByWorkbenchId.get(newest.workbenchId) ?? 0; + let newestFirstForPreview: readonly RoomMessage[] = [newest]; + if (isFailurePreviewParts(newest.parts)) { + const recentRows = await db + .select() + .from(workbenchMessages) + .where( + and( + inTenant, + eq(workbenchMessages.workbenchId, newest.workbenchId), + ), + ) + .orderBy( + desc(workbenchMessages.createdAt), + desc(workbenchMessages.id), + ) + .limit(PREVIEW_LOOKBACK); + newestFirstForPreview = recentRows.map((recent) => + toRoomMessage(recent as MessageRow), + ); + } result[newest.workbenchId] = summaryOf( newest, - unreadByWorkbenchId.get(newest.workbenchId) ?? 0, + unreadCount, + newestFirstForPreview, ); } return result; @@ -451,7 +527,18 @@ export function createInMemoryRoomMessageStore(): RoomMessageStore { workbench.sinceCreatedAt === undefined || message.createdAt > workbench.sinceCreatedAt, ).length; - result[workbench.workbenchId] = summaryOf(newest, unreadCount); + const newestFirst = [...messages] + .sort((left, right) => + left.createdAt === right.createdAt + ? right.id.localeCompare(left.id) + : right.createdAt.localeCompare(left.createdAt), + ) + .slice(0, PREVIEW_LOOKBACK); + result[workbench.workbenchId] = summaryOf( + newest, + unreadCount, + newestFirst, + ); } return result; },