From 10606c41e213b179d63a5e478881fda41ff3c095 Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Thu, 20 Aug 2026 22:22:28 -0700 Subject: [PATCH 1/3] Add tests for legible tool activity in the conversation Covers the presentation of a turn's tool calls end to end: the sentence a call renders as (no tool identifier, no argument JSON), the plain-text detail it opens onto, a failure that says why, a call still running in the present tense, and a run of consecutive calls folded into one round. The existing live-strip suites move to the same idiom: a tool call now asserts on the phrase a reader sees rather than on the raw tool name, and tool.done carries whether the call actually worked. --- .../chat-ui/src/agent-part-adapter.test.ts | 60 +--- packages/chat-ui/src/tool-activity.test.ts | 293 ++++++++++++++++++ packages/chat-ui/src/turn-activity.test.ts | 72 +++-- packages/chat-ui/test/components.test.tsx | 13 +- .../chat-ui/test/tool-activity-view.test.tsx | 141 +++++++++ .../chat-ui/test/use-turn-activity.test.tsx | 47 +-- 6 files changed, 510 insertions(+), 116 deletions(-) create mode 100644 packages/chat-ui/src/tool-activity.test.ts create mode 100644 packages/chat-ui/test/tool-activity-view.test.tsx diff --git a/packages/chat-ui/src/agent-part-adapter.test.ts b/packages/chat-ui/src/agent-part-adapter.test.ts index c9b301e77..78a4468ff 100644 --- a/packages/chat-ui/src/agent-part-adapter.test.ts +++ b/packages/chat-ui/src/agent-part-adapter.test.ts @@ -1,63 +1,5 @@ import { describe, expect, test } from "bun:test"; -import { toReactUiToolTrace, toReactUiReasoning } from "./agent-part-adapter"; - -describe("toReactUiToolTrace (CL-6318: chat's 4 statuses onto react-ui's 6)", () => { - const base = { - kind: "tool-trace" as const, - name: "slack__post", - input: { a: 1 }, - }; - - test("success is react-ui's output-available, the state that carries a result", () => { - const adapted = toReactUiToolTrace({ - ...base, - status: "success", - output: "ok", - }); - expect(adapted.status).toBe("output-available"); - expect(adapted.output).toBe("ok"); - }); - - test("pending, running and error carry across unchanged", () => { - for (const status of ["pending", "running", "error"] as const) { - expect(toReactUiToolTrace({ ...base, status }).status).toBe(status); - } - }); - - test("name and input ride along so the expanded detail can show them", () => { - const adapted = toReactUiToolTrace({ ...base, status: "running" }); - expect(adapted.name).toBe("slack__post"); - expect(adapted.input).toEqual({ a: 1 }); - }); - - test("the caller's message-scoped key becomes the toolCallId", () => { - // react-ui keys tool calls by id; chat's wire has no such field, so a - // key the caller already guarantees unique stands in rather than an - // invented id that could collide across messages. - const adapted = toReactUiToolTrace( - { ...base, status: "running" }, - "msg_1-3", - ); - expect(adapted.toolCallId).toBe("msg_1-3"); - }); - - test("output stays absent when the wire carried none", () => { - expect("output" in toReactUiToolTrace({ ...base, status: "running" })).toBe( - false, - ); - }); - - // The two approval states have no `@corbits/chat` equivalent — approvals - // ride as a `block` part bound to the real approvals flow, not as a tool - // status. Nothing should silently map onto them. - test("no chat status maps onto the approval states", () => { - const reachable = (["pending", "running", "success", "error"] as const).map( - (status) => toReactUiToolTrace({ ...base, status }).status, - ); - expect(reachable).not.toContain("approval-requested"); - expect(reachable).not.toContain("output-denied"); - }); -}); +import { toReactUiReasoning } from "./agent-part-adapter"; describe("toReactUiReasoning", () => { test("carries the text through", () => { diff --git a/packages/chat-ui/src/tool-activity.test.ts b/packages/chat-ui/src/tool-activity.test.ts new file mode 100644 index 000000000..f9aef125a --- /dev/null +++ b/packages/chat-ui/src/tool-activity.test.ts @@ -0,0 +1,293 @@ +// The rule this suite exists to hold: nothing a tool call produces reaches +// a reader as an identifier or as JSON. Every assertion below is either +// "this reads like a sentence" or "this string never appears". +import { describe, expect, test } from "bun:test"; +import type { Part, ToolTracePart } from "@corbits/chat/parts"; + +import { + describeToolCall, + describeToolRound, + groupTimelineParts, + plainTextOfOutput, + resolveToolIdentity, + summarizeToolOutput, + toToolActivityRow, +} from "./tool-activity"; + +function trace(part: Partial): ToolTracePart { + return { + kind: "tool-trace", + name: "search", + input: {}, + status: "success", + ...part, + } as ToolTracePart; +} + +describe("resolveToolIdentity", () => { + test("reads the real tool out of the generic MCP dispatch call", () => { + expect( + resolveToolIdentity("mcp_read", { + server: "notion", + tool: "search_pages", + }), + ).toEqual({ provider: "notion", words: ["search", "pages"] }); + }); + + test("splits a provider-namespaced tool on its double underscore", () => { + expect(resolveToolIdentity("slack__post_message", {})).toEqual({ + provider: "slack", + words: ["post", "message"], + }); + }); + + test("splits a dotted tool name too", () => { + expect(resolveToolIdentity("linear.save_issue", {})).toEqual({ + provider: "linear", + words: ["save", "issue"], + }); + }); + + test("a bare tool name has no provider", () => { + expect(resolveToolIdentity("webSearch", {})).toEqual({ + provider: undefined, + words: ["web", "search"], + }); + }); + + test("an MCP dispatch call with a malformed argument bag still resolves", () => { + expect(resolveToolIdentity("mcp_read", { server: "notion" })).toEqual({ + provider: undefined, + words: ["mcp", "read"], + }); + }); +}); + +describe("describeToolCall", () => { + test("a web search names what was searched for, in the right tense", () => { + expect( + describeToolCall("web_search", { query: "bench pricing" }, "past"), + ).toBe('Searched the web for "bench pricing"'); + expect( + describeToolCall("web_search", { query: "bench pricing" }, "present"), + ).toBe('Searching the web for "bench pricing"'); + }); + + test("a provider tool reads as a sentence naming the provider, not the identifier", () => { + const phrase = describeToolCall( + "slack__post_message", + { channel: "general" }, + "past", + ); + expect(phrase).toBe("Posted a message in Slack #general"); + expect(phrase).not.toContain("__"); + expect(phrase).not.toContain("slack__post_message"); + }); + + test("a file tool names the file, not its full path", () => { + expect( + describeToolCall( + "write_file", + { path: "/srv/app/src/report.md" }, + "past", + ), + ).toBe("Wrote a file — report.md"); + }); + + test("a url tool names the host", () => { + expect( + describeToolCall( + "fetch_page", + { url: "https://www.example.com/a/b" }, + "past", + ), + ).toBe("Fetched a page on example.com"); + }); + + test("an MCP dispatch call describes the tool it actually invoked", () => { + expect( + describeToolCall( + "mcp_call", + { server: "linear", tool: "save_issue", query: "auth bug" }, + "past", + ), + ).toBe('Saved an issue in Linear for "auth bug"'); + }); + + test("an unknown verb still loses its underscores rather than leaking raw", () => { + const phrase = describeToolCall("acme__frobnicate_widget", {}, "past"); + expect(phrase).toBe("Frobnicate widget in Acme"); + expect(phrase).not.toContain("_"); + }); + + test("a plural object drops the article", () => { + expect(describeToolCall("list_files", {}, "past")).toBe("Listed files"); + }); +}); + +describe("plainTextOfOutput", () => { + test("pulls the text out of MCP content blocks", () => { + expect( + plainTextOfOutput([ + { type: "text", text: "first" }, + { type: "text", text: "second" }, + ]), + ).toBe("first\nsecond"); + }); + + test("unwraps a result envelope's content", () => { + expect(plainTextOfOutput({ content: [{ type: "text", text: "hi" }] })).toBe( + "hi", + ); + }); + + test("returns nothing for an opaque object rather than stringifying it", () => { + expect(plainTextOfOutput({ rows: 4, cursor: "abc" })).toBeUndefined(); + }); +}); + +describe("summarizeToolOutput", () => { + test("a failure always says something, even with nothing to go on", () => { + expect(summarizeToolOutput("failed", undefined)).toBe("No reason given."); + }); + + test("a failure keeps its first line as the reason", () => { + expect( + summarizeToolOutput("failed", "Repository not found\n at listIssues"), + ).toBe("Repository not found"); + }); + + test("a success with no prose falls back to counting the results", () => { + expect(summarizeToolOutput("success", [{ id: 1 }, { id: 2 }])).toBe( + "2 results.", + ); + expect(summarizeToolOutput("success", [])).toBe("Nothing found."); + }); + + test("a running call has no detail to open onto yet", () => { + expect(summarizeToolOutput("running", undefined)).toBeUndefined(); + }); + + test("an opaque success detail never becomes JSON", () => { + const detail = summarizeToolOutput("success", { cursor: "abc" }); + expect(detail).toBeUndefined(); + }); +}); + +describe("toToolActivityRow", () => { + test("a failed call is failed, and says why in plain text", () => { + const row = toToolActivityRow( + trace({ + name: "github__get_issue", + input: { repo: "corbitsdev/workbench" }, + status: "error", + output: [{ type: "text", text: "Repository not found" }], + }), + "k", + ); + expect(row.status).toBe("failed"); + expect(row.phrase).toBe( + "Retrieved an issue in GitHub corbitsdev/workbench", + ); + expect(row.detail).toBe("Repository not found"); + }); + + test("a running call speaks in the present tense", () => { + const row = toToolActivityRow( + trace({ name: "web_search", input: { query: "x" }, status: "running" }), + "k", + ); + expect(row.phrase).toBe('Searching the web for "x"'); + expect(row.detail).toBeUndefined(); + }); +}); + +describe("describeToolRound", () => { + const settled = (status: "success" | "failed", phrase: string) => ({ + key: phrase, + toolName: "t", + phrase, + detail: undefined, + status, + }); + + test("a round still working speaks as the step that is working", () => { + const round = describeToolRound([ + settled("success", "Read a file"), + { + key: "b", + toolName: "t", + phrase: "Searching the web", + detail: undefined, + status: "running" as const, + }, + ]); + expect(round.label).toBe("Searching the web"); + expect(round.status).toBe("running"); + expect(round.opensByDefault).toBe(false); + }); + + test("a settled round counts its steps and stays closed", () => { + const round = describeToolRound([ + settled("success", "a"), + settled("success", "b"), + settled("success", "c"), + ]); + expect(round.label).toBe("3 steps"); + expect(round.opensByDefault).toBe(false); + }); + + test("a round with failures says so plainly and opens itself", () => { + const round = describeToolRound([ + settled("success", "a"), + settled("failed", "b"), + settled("failed", "c"), + ]); + expect(round.label).toBe("3 steps, 2 didn't work"); + expect(round.status).toBe("failed"); + expect(round.opensByDefault).toBe(true); + }); +}); + +describe("groupTimelineParts", () => { + const text = (value: string): Part => ({ kind: "text", text: value }); + + test("consecutive tool calls fold into one round", () => { + const groups = groupTimelineParts( + [ + text("Looking into it."), + trace({ name: "web_search" }), + trace({ name: "read_file" }), + trace({ name: "write_file" }), + text("Here is what I found."), + ], + "m1", + ); + expect(groups.map((group) => group.kind)).toEqual([ + "part", + "tool-activity", + "part", + ]); + const round = groups[1]; + expect(round?.kind === "tool-activity" && round.rows.length).toBe(3); + }); + + test("rounds separated by prose stay separate rounds", () => { + const groups = groupTimelineParts( + [trace({}), text("mid"), trace({}), trace({})], + "m1", + ); + expect(groups.map((group) => group.kind)).toEqual([ + "tool-activity", + "part", + "tool-activity", + ]); + }); + + test("a message with no tool calls is unchanged", () => { + const groups = groupTimelineParts([text("hello")], "m1"); + expect(groups).toEqual([ + { kind: "part", part: text("hello"), key: "m1-0" }, + ]); + }); +}); diff --git a/packages/chat-ui/src/turn-activity.test.ts b/packages/chat-ui/src/turn-activity.test.ts index fb70ac9f5..d8b1c9a56 100644 --- a/packages/chat-ui/src/turn-activity.test.ts +++ b/packages/chat-ui/src/turn-activity.test.ts @@ -1,6 +1,6 @@ import { describe, expect, test } from "bun:test"; -import { friendlyToolLabel, nextTurnActivityState } from "./turn-activity"; +import { nextTurnActivityState } from "./turn-activity"; function agentEvent(inner: unknown) { return { eventType: "chat.agent", data: inner }; @@ -58,7 +58,7 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", }); describe("tool call lifecycle", () => { - test("inference.tool_call.start opens a running chip named after the tool", () => { + test("inference.tool_call.start opens a running row for the tool", () => { const state = nextTurnActivityState( null, agentEvent({ @@ -71,7 +71,8 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", expect(state?.toolCalls).toEqual([ { callId: "c1", - label: "web_search", + name: "web_search", + input: undefined, status: "running", startedAtMs: 1000, doneAtMs: null, @@ -79,7 +80,7 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", ]); }); - test("inference.tool_call.end refines the label (e.g. mcp_read -> server.tool) without resetting startedAtMs", () => { + test("inference.tool_call.end attaches the call's arguments without resetting startedAtMs", () => { let state = nextTurnActivityState( null, agentEvent({ @@ -106,7 +107,8 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", expect(state?.toolCalls).toEqual([ { callId: "c1", - label: "notion.search_pages", + name: "mcp_read", + input: { server: "notion", tool: "search_pages" }, status: "running", startedAtMs: 1000, doneAtMs: null, @@ -114,7 +116,7 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", ]); }); - test("tool.start opens a chip directly (no preceding inference.tool_call.start) using the ToolCall's own arguments", () => { + test("tool.start opens a row directly (no preceding inference.tool_call.start) using the ToolCall's own arguments", () => { const state = nextTurnActivityState( null, agentEvent({ @@ -133,7 +135,8 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", expect(state?.toolCalls).toEqual([ { callId: "c2", - label: "linear.save_issue", + name: "mcp_call", + input: { server: "linear", tool: "save_issue" }, status: "running", startedAtMs: 500, doneAtMs: null, @@ -141,7 +144,7 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", ]); }); - test("tool.done marks the matching chip done and freezes its elapsed time", () => { + test("tool.done settles the matching row as a success and freezes its elapsed time", () => { let state = nextTurnActivityState( null, agentEvent({ @@ -163,14 +166,41 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", expect(state?.toolCalls).toEqual([ { callId: "c1", - label: "search", - status: "done", + name: "search", + input: {}, + status: "success", startedAtMs: 1000, doneAtMs: 4000, }, ]); }); + test("tool.done with isError settles the row as failed, not as a quiet success", () => { + let state = nextTurnActivityState( + null, + agentEvent({ + type: "tool.start", + seq: 1, + data: { + call: { id: "c1", name: "github__get_issue", arguments: {} }, + }, + }), + 1000, + ); + state = nextTurnActivityState( + state, + agentEvent({ + type: "tool.done", + seq: 2, + data: { + result: { callId: "c1", content: "not found", isError: true }, + }, + }), + 2000, + ); + expect(state?.toolCalls[0]?.status).toBe("failed"); + }); + test("tool.done for an unknown callId is ignored rather than crashing", () => { const state = nextTurnActivityState( null, @@ -283,25 +313,3 @@ describe("nextTurnActivityState (CL-6196: live tool-call and thinking states)", }); }); }); - -describe("friendlyToolLabel", () => { - test("an ordinary tool name is shown as-is", () => { - expect(friendlyToolLabel("web_search", undefined)).toBe("web_search"); - }); - - test("mcp_read/mcp_call resolve to server.tool when arguments carry both", () => { - expect( - friendlyToolLabel("mcp_read", { server: "notion", tool: "search" }), - ).toBe("notion.search"); - expect( - friendlyToolLabel("mcp_call", { server: "linear", tool: "save_issue" }), - ).toBe("linear.save_issue"); - }); - - test("mcp_read/mcp_call fall back to the bare tool name when arguments are missing or malformed", () => { - expect(friendlyToolLabel("mcp_read", undefined)).toBe("mcp_read"); - expect(friendlyToolLabel("mcp_call", { server: "notion" })).toBe( - "mcp_call", - ); - }); -}); diff --git a/packages/chat-ui/test/components.test.tsx b/packages/chat-ui/test/components.test.tsx index 1ab0c7e6a..48b124463 100644 --- a/packages/chat-ui/test/components.test.tsx +++ b/packages/chat-ui/test/components.test.tsx @@ -66,12 +66,11 @@ describe("WorkbenchTimeline", () => { // CL-6318: the timeline used to render text, event, file and block and // drop `reasoning` and `tool-trace` on the floor — the agent's thinking // and every tool call it made were invisible in the product. - test("renders a tool call, naming the tool and marking it finished", () => { + test("renders a tool call as a sentence, with the tool id nowhere in sight", () => { const markup = renderToStaticMarkup(); - expect(markup).toContain('data-slot="tool-block"'); - // react-ui presents the raw tool id in human phrasing. - expect(markup).toContain("Search"); - expect(markup).toContain('data-status="output-available"'); + expect(markup).toContain("chat-tool-activity"); + expect(markup).toContain("Searched for "x""); + expect(markup).toContain('data-status="success"'); }); test("renders reasoning as a disclosure rather than dropping it", () => { @@ -107,9 +106,9 @@ describe("WorkbenchTimeline", () => { }, ]; const markup = renderToStaticMarkup(); - expect(markup).toContain("Deploy"); + expect(markup).toContain("Deploying"); expect(markup).toContain('data-status="running"'); - expect(markup).not.toContain('data-status="output-available"'); + expect(markup).not.toContain('data-status="success"'); }); test("shows the sender's name when present", () => { diff --git a/packages/chat-ui/test/tool-activity-view.test.tsx b/packages/chat-ui/test/tool-activity-view.test.tsx new file mode 100644 index 000000000..2564a4e00 --- /dev/null +++ b/packages/chat-ui/test/tool-activity-view.test.tsx @@ -0,0 +1,141 @@ +// What a turn's tool activity actually renders: mounted for real, asserted +// on the text a reader sees. The standing rule under test is that no +// argument bag or tool result ever reaches the DOM as JSON, in any state. +import { beforeEach, describe, expect, test } from "bun:test"; +import { act } from "react"; +import { createRoot, type Root } from "react-dom/client"; + +import type { ToolTracePart } from "@corbits/chat/parts"; +import { toToolActivityRow } from "../src/tool-activity"; +import { ToolActivityGroup } from "../src/tool-activity-view"; + +function trace(part: Partial): ToolTracePart { + return { + kind: "tool-trace", + name: "web_search", + input: {}, + status: "success", + ...part, + } as ToolTracePart; +} + +function mount(parts: readonly ToolTracePart[]): HTMLDivElement { + const container = document.createElement("div"); + document.body.appendChild(container); + const root: Root = createRoot(container); + const rows = parts.map((part, index) => toToolActivityRow(part, `k${index}`)); + act(() => { + root.render(); + }); + return container; +} + +function click(element: Element | null) { + act(() => { + (element as HTMLElement).click(); + }); +} + +beforeEach(() => { + document.body.innerHTML = ""; +}); + +describe("ToolActivityGroup", () => { + test("a successful call reads as a sentence, with no identifier or JSON", () => { + const el = mount([ + trace({ + name: "slack__post_message", + input: { channel: "general", text: "shipping now" }, + output: [{ type: "text", text: "delivered" }], + }), + ]); + expect(el.textContent).toContain("Posted a message in Slack #general"); + expect(el.textContent).not.toContain("slack__post_message"); + expect(el.textContent).not.toContain("{"); + expect(el.textContent).not.toContain('"channel"'); + }); + + test("detail stays closed until asked for, then shows plain text", () => { + const el = mount([ + trace({ + name: "web_search", + input: { query: "bench pricing" }, + output: [{ type: "text", text: "Eight matching pages." }], + }), + ]); + expect(el.textContent).not.toContain("Eight matching pages."); + click(el.querySelector(".chat-tool-activity-trigger")); + expect(el.textContent).toContain("Eight matching pages."); + }); + + test("a failed call says so plainly and opens onto the reason", () => { + const el = mount([ + trace({ + name: "github__get_issue", + input: { repo: "corbitsdev/workbench" }, + status: "error", + output: [{ type: "text", text: "Repository not found" }], + }), + ]); + const marker = el.querySelector(".chat-tool-activity-marker"); + expect(marker?.getAttribute("data-status")).toBe("failed"); + click(el.querySelector(".chat-tool-activity-trigger")); + expect(el.textContent).toContain("Repository not found"); + }); + + test("a failure with nothing to say still says that much", () => { + const el = mount([trace({ name: "run_command", status: "error" })]); + click(el.querySelector(".chat-tool-activity-trigger")); + expect(el.textContent).toContain("No reason given."); + }); + + test("a call still running speaks in the present tense and offers no disclosure", () => { + const el = mount([ + trace({ name: "web_search", input: { query: "x" }, status: "running" }), + ]); + expect(el.textContent).toContain('Searching the web for "x"'); + expect(el.querySelector(".chat-tool-activity-trigger")).toBeNull(); + expect( + el + .querySelector(".chat-tool-activity-marker") + ?.getAttribute("data-status"), + ).toBe("running"); + }); + + test("consecutive rounds collapse to one line that opens onto the steps", () => { + const el = mount([ + trace({ name: "web_search", input: { query: "a" } }), + trace({ name: "read_file", input: { path: "src/app.ts" } }), + trace({ name: "write_file", input: { path: "src/app.ts" } }), + ]); + expect(el.textContent).toContain("3 steps"); + expect(el.textContent).not.toContain("Wrote a file"); + + click(el.querySelector(".chat-tool-activity-trigger")); + expect(el.textContent).toContain('Searched the web for "a"'); + expect(el.textContent).toContain("Read a file — app.ts"); + expect(el.textContent).toContain("Wrote a file — app.ts"); + }); + + test("a round that is still working shows the step that is working", () => { + const el = mount([ + trace({ name: "read_file", input: { path: "a.ts" } }), + trace({ name: "web_search", input: { query: "b" }, status: "running" }), + ]); + expect(el.textContent).toContain('Searching the web for "b"'); + expect(el.textContent).not.toContain("2 steps"); + }); + + test("a round containing a failure names the failure and opens itself", () => { + const el = mount([ + trace({ name: "read_file", input: { path: "a.ts" } }), + trace({ + name: "github__get_issue", + status: "error", + output: "Repository not found", + }), + ]); + expect(el.textContent).toContain("2 steps, 1 didn't work"); + expect(el.textContent).toContain("Retrieved an issue in GitHub"); + }); +}); diff --git a/packages/chat-ui/test/use-turn-activity.test.tsx b/packages/chat-ui/test/use-turn-activity.test.tsx index d7fba4c03..3370e3782 100644 --- a/packages/chat-ui/test/use-turn-activity.test.tsx +++ b/packages/chat-ui/test/use-turn-activity.test.tsx @@ -51,9 +51,9 @@ function mount(initialWorkbenchId: string | null, staleMs?: number) { } describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { - test("a tool call in flight renders a running chip with its friendly label", () => { + test("a tool call in flight renders a running row phrased in the present tense", () => { const harness = mount("chan_a"); - expect(harness.container.querySelector(".chat-turn-activity")).toBeNull(); + expect(harness.container.querySelector(".chat-tool-activity")).toBeNull(); harness.send("chat.agent", { type: "inference.tool_call.start", @@ -62,15 +62,20 @@ describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { }); const row = harness.container.querySelector( - ".chat-turn-activity-row", + ".chat-tool-activity-row", ) as HTMLElement; expect(row).not.toBeNull(); - expect(row.dataset.status).toBe("running"); - expect(row.textContent).toContain("web_search"); + expect( + row + .querySelector(".chat-tool-activity-marker") + ?.getAttribute("data-status"), + ).toBe("running"); + expect(row.textContent).toContain("Searching the web"); + expect(row.textContent).not.toContain("web_search"); harness.unmount(); }); - test("mcp_read resolves to server.tool once inference.tool_call.end arrives", () => { + test("an MCP dispatch call reads as the tool it actually invoked once its arguments arrive", () => { const harness = mount("chan_a"); harness.send("chat.agent", { type: "inference.tool_call.start", @@ -88,11 +93,13 @@ describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { }, }); - expect(harness.container.textContent).toContain("notion.search_pages"); + expect(harness.container.textContent).toContain( + "Searching pages in Notion", + ); harness.unmount(); }); - test("tool.done flips the chip to done, and the strip disappears once the turn ends", () => { + test("tool.done settles the row, and the strip disappears once the turn ends", () => { const harness = mount("chan_a"); harness.send("chat.agent", { type: "tool.start", @@ -106,16 +113,20 @@ describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { }); const row = harness.container.querySelector( - ".chat-turn-activity-row", + ".chat-tool-activity-row", ) as HTMLElement; - expect(row.dataset.status).toBe("done"); + expect( + row + .querySelector(".chat-tool-activity-marker") + ?.getAttribute("data-status"), + ).toBe("success"); harness.send("chat.agent", { type: "inference.done", seq: 3, data: { turn: {}, usage: {}, source: "primary" }, }); - expect(harness.container.querySelector(".chat-turn-activity")).toBeNull(); + expect(harness.container.querySelector(".chat-tool-activity")).toBeNull(); harness.unmount(); }); @@ -128,7 +139,7 @@ describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { }); expect( - harness.container.querySelector(".chat-turn-activity-thinking"), + harness.container.querySelector(".chat-tool-activity-thinking"), ).not.toBeNull(); harness.unmount(); }); @@ -141,11 +152,11 @@ describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { data: { call: { id: "c1", name: "search", arguments: {} } }, }); expect( - harness.container.querySelector(".chat-turn-activity-row"), + harness.container.querySelector(".chat-tool-activity-row"), ).not.toBeNull(); harness.switchWorkbench("chan_b"); - expect(harness.container.querySelector(".chat-turn-activity")).toBeNull(); + expect(harness.container.querySelector(".chat-tool-activity")).toBeNull(); harness.unmount(); }); @@ -157,12 +168,12 @@ describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { data: { call: { id: "c1", name: "search", arguments: {} } }, }); expect( - harness.container.querySelector(".chat-turn-activity"), + harness.container.querySelector(".chat-tool-activity"), ).not.toBeNull(); // No `tool.done`/`reactor.done` ever arrives — a dropped SSE mid-turn. await harness.settle(60); - expect(harness.container.querySelector(".chat-turn-activity")).toBeNull(); + expect(harness.container.querySelector(".chat-tool-activity")).toBeNull(); harness.unmount(); }); @@ -182,11 +193,11 @@ describe("useTurnActivity + TurnActivityStrip (CL-6196: live wiring)", () => { }); await harness.settle(20); expect( - harness.container.querySelector(".chat-turn-activity"), + harness.container.querySelector(".chat-tool-activity"), ).not.toBeNull(); await harness.settle(30); - expect(harness.container.querySelector(".chat-turn-activity")).toBeNull(); + expect(harness.container.querySelector(".chat-tool-activity")).toBeNull(); harness.unmount(); }); }); From 6e393559a2a7a71a95ed54528033f77f63b32c84 Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Thu, 20 Aug 2026 22:22:50 -0700 Subject: [PATCH 2/3] Chat: render tool activity as legible rows, not tool ids and JSON MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A turn's tool calls used to reach the reader as raw material. In the transcript each call rendered as its own block titled with the tool's identifier run through a humaniser ("Mcp read", "Post message (Slack)"), and opening one showed its arguments and its result as JSON.stringify output. Five consecutive calls stacked as five blocks between the question and the answer. In the live strip a call showed the bare tool name, and a call that failed settled into the same quiet marker as one that succeeded — a failure was invisible. Now one vocabulary serves both. tool-activity.ts turns a call into a sentence in the tense its status calls for ("Searching the web for 'pricing'" while it runs, "Searched the web for 'pricing'" once it settles), resolving the generic MCP dispatch tools to the tool they actually invoked, and turns a result into plain text — prose when the tool returned any, a count when it returned a list, and never JSON. A failure always says something, "No reason given." when the tool offered nothing. Consecutive calls fold into one round that collapses to a single line: the step currently working while the turn is open, then a count once settled, saying plainly if any of them didn't work. A round with a failure opens itself; everything else stays closed, so the reply stays above its own machinery. The live strip and the transcript now render through the same rows, so a call does not restyle itself the moment the turn ends. react-ui's ToolBlock is no longer used here: it renders one call at a time and stringifies its arguments and output. --- packages/chat-ui/src/agent-part-adapter.ts | 42 +- packages/chat-ui/src/styles.css | 115 ++++- packages/chat-ui/src/timeline.tsx | 34 +- packages/chat-ui/src/tool-activity-view.tsx | 165 +++++++ packages/chat-ui/src/tool-activity.ts | 491 ++++++++++++++++++++ packages/chat-ui/src/turn-activity.tsx | 183 ++++---- 6 files changed, 876 insertions(+), 154 deletions(-) create mode 100644 packages/chat-ui/src/tool-activity-view.tsx create mode 100644 packages/chat-ui/src/tool-activity.ts diff --git a/packages/chat-ui/src/agent-part-adapter.ts b/packages/chat-ui/src/agent-part-adapter.ts index fdfa1f728..ed0e975c4 100644 --- a/packages/chat-ui/src/agent-part-adapter.ts +++ b/packages/chat-ui/src/agent-part-adapter.ts @@ -1,42 +1,14 @@ // Adapting `@corbits/chat`'s parts to the shapes `@corbits/react-ui` // renders (CL-6318). // -// The two unions agree on their kinds — text, reasoning, tool-trace, -// block, file — because react-ui's was built against this wire. They do -// not agree on the tool-call status enum: chat's has four states, -// react-ui's has six, adding `approval-requested` and `output-denied`. -// Those two have no chat equivalent today (an approval rides as a `block` -// part bound to the real approvals flow, not as a tool status), so nothing -// here maps onto them — a gap to close upstream-first if the wire ever -// grows them, never by inventing a status the server never sent. +// Only reasoning crosses this boundary now. Tool calls used to as well, +// but the conversation renders them through `tool-activity.tsx` instead: +// react-ui's `ToolBlock` shows one call at a time with its arguments and +// its result as `JSON.stringify` output, and the chat surface groups a +// turn's calls into rounds and never shows a reader JSON at all. -import type { - PartReasoning, - PartToolTrace, - ToolTraceStatus, -} from "@corbits/react-ui"; -import type { ReasoningPart, ToolTracePart } from "@corbits/chat/parts"; - -/** chat's `success` is react-ui's `output-available` — the same state - * under the name that says what it carries. The rest are identical. */ -export function toReactUiToolTrace( - part: ToolTracePart, - /** react-ui keys tool calls by id; chat's wire carries none, so the - * caller passes its own message-scoped key rather than an invented id - * that could collide across messages. */ - toolCallId = "", -): PartToolTrace { - const status: ToolTraceStatus = - part.status === "success" ? "output-available" : part.status; - const base = { - kind: "tool-trace" as const, - toolCallId, - name: part.name, - status, - input: part.input, - }; - return part.output !== undefined ? { ...base, output: part.output } : base; -} +import type { PartReasoning } from "@corbits/react-ui"; +import type { ReasoningPart } from "@corbits/chat/parts"; /** chat's reasoning carries no duration; react-ui's is optional, so it * stays absent rather than being fabricated. */ diff --git a/packages/chat-ui/src/styles.css b/packages/chat-ui/src/styles.css index 9f5f436d3..3c5fc4c8c 100644 --- a/packages/chat-ui/src/styles.css +++ b/packages/chat-ui/src/styles.css @@ -1959,53 +1959,138 @@ row, and a retry note, shown above the composer while a turn streams. Square, zero-radius, hairline borders — matches `.chat-block`'s generative-UI language rather than the rounded typing-dot pill. */ -.chat-turn-activity { +.chat-tool-activity { display: flex; flex-direction: column; + gap: 0.15rem; + padding: 0.15rem 0; + font-size: 0.72rem; + color: var(--muted-foreground); +} + +.chat-tool-activity-live { gap: 0.3rem; padding: 0.35rem 0.7rem; border-top: 1px solid color-mix(in srgb, var(--foreground) 10%, transparent); } -.chat-turn-activity-row { +.chat-tool-activity-row { display: flex; + flex-direction: column; + gap: 0.2rem; +} + +.chat-tool-activity-row[data-indented="true"] { + padding-left: 0.85rem; + border-left: 1px solid color-mix(in srgb, var(--foreground) 10%, transparent); + margin-left: 0.18rem; +} + +/* A row with no detail is inert text, not a control: it lines up with the + triggers above and below it but never invites a click. */ +.chat-tool-activity-row:not(:has(.chat-tool-activity-trigger)) { + flex-direction: row; align-items: center; gap: 0.45rem; - font-size: 0.72rem; - color: var(--muted-foreground); + padding-block: 0.15rem; +} + +.chat-tool-activity-trigger { + display: flex; + align-items: center; + gap: 0.45rem; + width: 100%; + padding: 0.15rem 0; + border: 0; + border-radius: 0; + background: none; + color: inherit; + font: inherit; + text-align: left; + cursor: pointer; +} + +.chat-tool-activity-trigger:hover { + color: var(--foreground); } -.chat-turn-activity-marker { +.chat-tool-activity-trigger:focus-visible { + outline: 2px solid var(--primary); + outline-offset: 2px; +} + +.chat-tool-activity-marker { width: 0.4rem; height: 0.4rem; flex: none; border-radius: 0; - background: var(--primary); + background: var(--muted-foreground); + opacity: 0.6; } -.chat-turn-activity-row[data-status="running"] .chat-turn-activity-marker { +.chat-tool-activity-marker[data-status="running"], +.chat-tool-activity-marker[data-status="pending"] { + background: var(--primary); + opacity: 1; animation: chat-block-pulse 1.6s ease infinite; } -.chat-turn-activity-row[data-status="done"] .chat-turn-activity-marker { - background: var(--muted-foreground); - opacity: 0.6; +/* A failure is the one state that earns colour here — everything else in + this strip is deliberately quiet chrome. */ +.chat-tool-activity-marker[data-status="failed"] { + background: var(--destructive, #b42318); + opacity: 1; +} + +.chat-tool-activity-row[data-status="failed"] .chat-tool-activity-phrase { + color: var(--destructive, #b42318); } -.chat-turn-activity-label { +.chat-tool-activity-phrase { + min-width: 0; + flex: 1 1 auto; overflow-wrap: anywhere; } -.chat-turn-activity-elapsed { - margin-left: auto; +.chat-tool-activity-meta { + flex: none; font-variant-numeric: tabular-nums; + opacity: 0.75; +} + +.chat-tool-activity-caret { + flex: none; + width: 0.75rem; + height: 0.75rem; + transition: transform 0.15s ease-out; +} + +.chat-tool-activity-caret[data-open="true"] { + transform: rotate(90deg); +} + +.chat-tool-activity-detail { + margin: 0 0 0.2rem 0.85rem; + padding-left: 0.7rem; + border-left: 1px solid color-mix(in srgb, var(--foreground) 12%, transparent); + max-height: 12rem; + overflow: auto; + white-space: pre-wrap; + overflow-wrap: anywhere; + color: var(--muted-foreground); +} + +.chat-tool-activity-rows { + display: flex; + flex-direction: column; + gap: 0.15rem; } -.chat-turn-activity-thinking { +.chat-tool-activity-thinking { font-style: italic; } -.chat-turn-activity-retry { +.chat-tool-activity-retry { color: var(--warn, #b7791f); } diff --git a/packages/chat-ui/src/timeline.tsx b/packages/chat-ui/src/timeline.tsx index 5eb420943..5b6873123 100644 --- a/packages/chat-ui/src/timeline.tsx +++ b/packages/chat-ui/src/timeline.tsx @@ -22,11 +22,11 @@ import { Button, EmptyState, PartsRenderer, - ToolBlock, toast, - toolTraceToBlockState, } from "@corbits/react-ui"; -import { toReactUiReasoning, toReactUiToolTrace } from "./agent-part-adapter"; +import { toReactUiReasoning } from "./agent-part-adapter"; +import { groupTimelineParts } from "./tool-activity"; +import { ToolActivityGroup } from "./tool-activity-view"; import { ArrowBendUpLeft, ChatCircle, @@ -1234,8 +1234,12 @@ function MessageParts({ > {showDayDivider && }
- {item.parts.map((part, index) => { - const key = `${groupKey}-${index}`; + {groupTimelineParts(item.parts, groupKey).map((group) => { + const key = group.key; + if (group.kind === "tool-activity") { + return ; + } + const part = group.part; if (part.kind === "text" && part.turnFailed === true) { return ( ); } - // The agent's own thinking and its tool calls (CL-6318). Both - // render through react-ui, which already owns this presentation — - // a reasoning disclosure and the tool-call lifecycle — so the - // workbench carries no second version of either. + // The agent's own thinking still renders through react-ui, which + // owns the reasoning disclosure. Tool calls no longer do: they + // arrive here already folded into rounds by `groupTimelineParts` + // above, and react-ui's `ToolBlock` renders one call at a time + // with its arguments and result as `JSON.stringify` output. if (part.kind === "reasoning") { return ( ); } - if (part.kind === "tool-trace") { - const trace = toReactUiToolTrace(part, key); - return ( - - ); - } if (part.kind === "block") { return (