From d5f65da078a04ac4edff9e50f67e3f13761223ad Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Sat, 29 Aug 2026 20:38:51 -0700 Subject: [PATCH 1/2] Parse inline Ollama tool JSON into tool calls Ollama/qwen-like models emit declared tool calls as JSON in content. The OpenAI-compatible parser only reads native tool_calls, so the director stopped with visible JSON instead of running the tool. --- packages/ollama-adapter/src/adapter.test.ts | 98 +++++- packages/ollama-adapter/src/adapter.ts | 32 +- packages/ollama-adapter/src/index.ts | 7 + .../src/inline-tool-json.test.ts | 193 ++++++++++++ .../ollama-adapter/src/inline-tool-json.ts | 293 ++++++++++++++++++ workflows/assistant/src/index.ts | 10 +- workflows/assistant/test/definition.test.ts | 13 + 7 files changed, 637 insertions(+), 9 deletions(-) create mode 100644 packages/ollama-adapter/src/inline-tool-json.test.ts create mode 100644 packages/ollama-adapter/src/inline-tool-json.ts diff --git a/packages/ollama-adapter/src/adapter.test.ts b/packages/ollama-adapter/src/adapter.test.ts index a4c009c6e..45ca88d53 100644 --- a/packages/ollama-adapter/src/adapter.test.ts +++ b/packages/ollama-adapter/src/adapter.test.ts @@ -1,7 +1,12 @@ import { describe, expect, test } from "bun:test"; import { createOpenAIAdapter } from "@intx/inference/providers"; import type { LastCycleSource } from "@intx/types/runtime"; -import type { ConversationTurn, InferenceOptions } from "@intx/types/runtime"; +import type { + ConversationTurn, + InferenceEvent, + InferenceOptions, + ToolDefinition, +} from "@intx/types/runtime"; import { createOllamaAdapter } from "./adapter"; @@ -21,10 +26,48 @@ const messages: ConversationTurn[] = [ const options: InferenceOptions = {}; +const memorySearchTool: ToolDefinition = { + name: "memory_search", + description: "Search firm memory", + inputSchema: { type: "object" }, +}; + +const CL_7186_PAYLOAD = + '{"name":"memory_search","parameters":{"query":"this person"}}'; + function bodyOf(built: { body: string }): Record { return JSON.parse(built.body) as Record; } +function toolStarts(events: readonly InferenceEvent[]): InferenceEvent[] { + return events.filter((event) => event.type === "inference.tool_call.start"); +} + +function textDeltas(events: readonly InferenceEvent[]): InferenceEvent[] { + return events.filter((event) => event.type === "inference.text.delta"); +} + +function argumentFragments(events: readonly InferenceEvent[]): string { + return events + .filter((event) => event.type === "inference.tool_call.delta") + .map( + (event) => (event.data as { argumentFragment: string }).argumentFragment, + ) + .join(""); +} + +function expectSharedToolCallId(events: readonly InferenceEvent[]): void { + const ids = events + .filter( + (event) => + event.type === "inference.tool_call.start" || + event.type === "inference.tool_call.delta", + ) + .map((event) => (event.data as { callId: string }).callId); + expect(ids.length).toBeGreaterThan(0); + expect(new Set(ids)).toEqual(new Set(["ollama-inline-0"])); +} + describe("createOllamaAdapter", () => { test("no override configured leaves the body equivalent to the built-in adapter's", () => { const wrapped = createOllamaAdapter(source, undefined); @@ -84,4 +127,57 @@ describe("createOllamaAdapter", () => { expect(typeof wrapped.extractRetryAfterMs).toBe("function"); expect(typeof wrapped.extractPacingDelayMs).toBe("function"); }); + + test("parseResponse salvages declared inline memory_search JSON into a tool call", () => { + const wrapped = createOllamaAdapter(source, undefined); + wrapped.buildRequest(messages, "gpt-oss:20b", { + tools: [memorySearchTool], + }); + const events = [ + ...wrapped.parseResponse( + JSON.stringify({ + choices: [{ delta: { content: CL_7186_PAYLOAD } }], + }), + ), + ...wrapped.parseResponse( + JSON.stringify({ + choices: [{ delta: {}, finish_reason: "stop" }], + }), + ), + ]; + expect(textDeltas(events)).toEqual([]); + expect((toolStarts(events)[0]?.data as { name: string }).name).toBe( + "memory_search", + ); + expectSharedToolCallId(events); + expect(JSON.parse(argumentFragments(events))).toEqual({ + query: "this person", + }); + }); + + test("parseJSONResponse salvages declared inline memory_search JSON into a tool call", () => { + const wrapped = createOllamaAdapter(source, undefined); + wrapped.buildRequest(messages, "gpt-oss:20b", { + tools: [memorySearchTool], + }); + const events = wrapped.parseJSONResponse( + JSON.stringify({ + object: "chat.completion", + choices: [ + { + message: { role: "assistant", content: CL_7186_PAYLOAD }, + }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1 }, + }), + ); + expect(textDeltas(events)).toEqual([]); + expect((toolStarts(events)[0]?.data as { name: string }).name).toBe( + "memory_search", + ); + expectSharedToolCallId(events); + expect(JSON.parse(argumentFragments(events))).toEqual({ + query: "this person", + }); + }); }); diff --git a/packages/ollama-adapter/src/adapter.ts b/packages/ollama-adapter/src/adapter.ts index 8a3bac82e..4eb2520dc 100644 --- a/packages/ollama-adapter/src/adapter.ts +++ b/packages/ollama-adapter/src/adapter.ts @@ -26,6 +26,12 @@ import { type OllamaAdapterOverride, } from "./overrides"; import { createThinkSplitState, reclassifyThinkingEvents } from "./think-tags"; +import { + createInlineToolJsonState, + reclassifyInlineToolJsonEvents, + responseChunkIsTerminal, + setDeclaredToolNames, +} from "./inline-tool-json"; type OllamaChatBody = { options?: Record; @@ -85,16 +91,32 @@ export const createOllamaAdapter: AdapterFactory = ( // chunk of one response without leaking state between requests. const streamThinkState = createThinkSplitState(); const jsonThinkState = createThinkSplitState(); + const streamInlineState = createInlineToolJsonState(); + const jsonInlineState = createInlineToolJsonState(); return { ...inner, - buildRequest: (messages, model, options) => - applyOverride( + buildRequest: (messages, model, options) => { + setDeclaredToolNames(streamInlineState, options.tools); + setDeclaredToolNames(jsonInlineState, options.tools); + return applyOverride( inner.buildRequest(messages, model, options), resolveOverride(config, model), - ), + ); + }, parseResponse: (sseData) => - reclassifyThinkingEvents(inner.parseResponse(sseData), streamThinkState), + reclassifyInlineToolJsonEvents( + reclassifyThinkingEvents( + inner.parseResponse(sseData), + streamThinkState, + ), + streamInlineState, + { flush: responseChunkIsTerminal(sseData) }, + ), parseJSONResponse: (body) => - reclassifyThinkingEvents(inner.parseJSONResponse(body), jsonThinkState), + reclassifyInlineToolJsonEvents( + reclassifyThinkingEvents(inner.parseJSONResponse(body), jsonThinkState), + jsonInlineState, + { flush: true }, + ), }; }; diff --git a/packages/ollama-adapter/src/index.ts b/packages/ollama-adapter/src/index.ts index d7bd14631..4b1393f8d 100644 --- a/packages/ollama-adapter/src/index.ts +++ b/packages/ollama-adapter/src/index.ts @@ -11,3 +11,10 @@ export { reclassifyThinkingEvents, type ThinkSplitState, } from "./think-tags"; +export { + createInlineToolJsonState, + reclassifyInlineToolJsonEvents, + responseChunkIsTerminal, + setDeclaredToolNames, + type InlineToolJsonState, +} from "./inline-tool-json"; diff --git a/packages/ollama-adapter/src/inline-tool-json.test.ts b/packages/ollama-adapter/src/inline-tool-json.test.ts new file mode 100644 index 000000000..3e131a740 --- /dev/null +++ b/packages/ollama-adapter/src/inline-tool-json.test.ts @@ -0,0 +1,193 @@ +import { describe, expect, test } from "bun:test"; +import type { InferenceEvent } from "@intx/types/runtime"; + +import { + createInlineToolJsonState, + reclassifyInlineToolJsonEvents, +} from "./inline-tool-json"; + +const CL_7186_PAYLOAD = + '{"name":"memory_search","parameters":{"query":"this person"}}'; + +function textDelta(token: string, seq = 1): InferenceEvent { + return { + type: "inference.text.delta", + seq, + data: { token, partial: { text: token }, index: 0 }, + }; +} + +function usageEvent(seq = 2): InferenceEvent { + return { + type: "inference.usage", + seq, + data: { + usage: { + input: 1, + output: 1, + cacheRead: 0, + cacheWrite: 0, + thinking: 0, + }, + source: { sourceId: "ollama-test", provider: "ollama", model: "qwen" }, + }, + }; +} + +function reclassify( + events: readonly InferenceEvent[], + declaredNames: Iterable, + opts?: { flush?: boolean }, +): InferenceEvent[] { + return reclassifyInlineToolJsonEvents( + events, + createInlineToolJsonState(declaredNames), + opts, + ); +} + +function toolStarts(events: readonly InferenceEvent[]): InferenceEvent[] { + return events.filter((event) => event.type === "inference.tool_call.start"); +} + +function textDeltas(events: readonly InferenceEvent[]): InferenceEvent[] { + return events.filter((event) => event.type === "inference.text.delta"); +} + +function toolCallIds(events: readonly InferenceEvent[]): string[] { + return events + .filter( + (event) => + event.type === "inference.tool_call.start" || + event.type === "inference.tool_call.delta", + ) + .map((event) => (event.data as { callId: string }).callId); +} + +function expectSharedToolCallId(events: readonly InferenceEvent[]): void { + const ids = toolCallIds(events); + expect(ids.length).toBeGreaterThan(0); + expect(new Set(ids)).toEqual(new Set(["ollama-inline-0"])); +} + +function argumentFragments(events: readonly InferenceEvent[]): string { + return events + .filter((event) => event.type === "inference.tool_call.delta") + .map( + (event) => (event.data as { argumentFragment: string }).argumentFragment, + ) + .join(""); +} + +describe("reclassifyInlineToolJsonEvents", () => { + test("the CL-7186 memory_search JSON becomes a tool_call.start with no text", () => { + const out = reclassify( + [textDelta(CL_7186_PAYLOAD), usageEvent()], + ["memory_search"], + ); + expect(textDeltas(out)).toEqual([]); + const starts = toolStarts(out); + expect(starts).toHaveLength(1); + expect((starts[0]?.data as { name: string }).name).toBe("memory_search"); + expectSharedToolCallId(out); + expect(JSON.parse(argumentFragments(out))).toEqual({ + query: "this person", + }); + }); + + test("chunks split across a JSON boundary still salvage once the object completes", () => { + const state = createInlineToolJsonState(["memory_search"]); + const splitAt = CL_7186_PAYLOAD.indexOf("memory_search") + 6; + const first = reclassifyInlineToolJsonEvents( + [textDelta(CL_7186_PAYLOAD.slice(0, splitAt))], + state, + ); + expect(textDeltas(first)).toEqual([]); + expect(toolStarts(first)).toEqual([]); + + const second = reclassifyInlineToolJsonEvents( + [textDelta(CL_7186_PAYLOAD.slice(splitAt))], + state, + ); + expect(textDeltas(second)).toEqual([]); + expect(toolStarts(second)).toEqual([]); + + const flushed = reclassifyInlineToolJsonEvents([], state, { flush: true }); + expect(textDeltas(flushed)).toEqual([]); + expect((toolStarts(flushed)[0]?.data as { name: string }).name).toBe( + "memory_search", + ); + expectSharedToolCallId(flushed); + expect(JSON.parse(argumentFragments(flushed))).toEqual({ + query: "this person", + }); + }); + + test("arguments is accepted as the args object, same as parameters", () => { + const payload = + '{"name":"memory_search","arguments":{"query":"this person"}}'; + const out = reclassify([textDelta(payload)], ["memory_search"], { + flush: true, + }); + expect(textDeltas(out)).toEqual([]); + expectSharedToolCallId(out); + expect(JSON.parse(argumentFragments(out))).toEqual({ + query: "this person", + }); + }); + + test("an unknown name stays text even when the JSON shape matches", () => { + const payload = + '{"name":"not_a_tool","parameters":{"query":"this person"}}'; + const original = [textDelta(payload)]; + const out = reclassify(original, ["memory_search"], { flush: true }); + expect(out).toEqual(original); + expect(toolStarts(out)).toEqual([]); + }); + + test("declared-name gate: no tools on the request leaves the JSON as text", () => { + const original = [textDelta(CL_7186_PAYLOAD)]; + const out = reclassify(original, [], { flush: true }); + expect(out).toEqual(original); + expect(toolStarts(out)).toEqual([]); + }); + + test("legitimate JSON answers stay text: extra keys, arrays, mixed prose, tutorials", () => { + const samples = [ + '{"name":"memory_search","parameters":{"query":"this person"},"title":"example"}', + '[{"name":"memory_search","parameters":{"query":"this person"}}]', + 'Call it like this:\n{"name":"memory_search","parameters":{"query":"this person"}}', + '{"ok":true,"count":3}', + '{"name":"memory_search"}', + "```json\n" + CL_7186_PAYLOAD + "\n```", + ]; + for (const sample of samples) { + const original = [textDelta(sample)]; + const out = reclassify(original, ["memory_search"], { flush: true }); + expect(out).toEqual(original); + expect(toolStarts(out)).toEqual([]); + } + }); + + test("incomplete JSON at flush stays the original text", () => { + const original = [textDelta('{"name":"memory_search"')]; + const out = reclassify(original, ["memory_search"], { flush: true }); + expect(out).toEqual(original); + }); + + test("ordinary prose with tools declared passes through unchanged", () => { + const original = [textDelta("hello, how can I help?")]; + const out = reclassify(original, ["memory_search"]); + expect(out).toEqual(original); + }); + + test("salvaged tool_call events reuse the text block index so they cannot collide with leftover content", () => { + const out = reclassify([textDelta(CL_7186_PAYLOAD)], ["memory_search"], { + flush: true, + }); + const start = toolStarts(out)[0]; + expect((start?.data as { index?: number }).index).toBe(0); + expectSharedToolCallId(out); + expect(textDeltas(out)).toEqual([]); + }); +}); diff --git a/packages/ollama-adapter/src/inline-tool-json.ts b/packages/ollama-adapter/src/inline-tool-json.ts new file mode 100644 index 000000000..0a6edc307 --- /dev/null +++ b/packages/ollama-adapter/src/inline-tool-json.ts @@ -0,0 +1,293 @@ +// Ollama's OpenAI-compatible endpoint sometimes emits a declared tool call +// as a JSON object in `content` instead of native `tool_calls` (CL-7186). +// `@intx/inference`'s OpenAI parser only reads `delta.tool_calls`, so that +// JSON rides downstream as assistant text and the tool never runs. This +// module reclassifies a content stream that is exactly that object into +// `inference.tool_call.*` events, gated on the tools declared for the +// request so unknown names and ordinary JSON answers stay text. +import type { InferenceEvent } from "@intx/types/runtime"; + +export type InlineToolJsonState = { + declaredNames: Set; + held: InferenceEvent[]; + acc: string; + verdict: "pending" | "text" | "tool"; +}; + +export function createInlineToolJsonState( + declaredNames: Iterable = [], +): InlineToolJsonState { + return { + declaredNames: new Set(declaredNames), + held: [], + acc: "", + verdict: "pending", + }; +} + +export function setDeclaredToolNames( + state: InlineToolJsonState, + tools: readonly { name: string }[] | undefined, +): void { + state.declaredNames = new Set((tools ?? []).map((tool) => tool.name)); +} + +const INLINE_TOOL_CALL_ID = "ollama-inline-0"; +const EMPTY_PARTIAL = { text: "" }; + +function isPlainObject(value: unknown): value is Record { + return ( + typeof value === "object" && + value !== null && + Array.isArray(value) === false + ); +} + +type ParseResult = + | { kind: "incomplete" } + | { kind: "reject" } + | { kind: "object"; value: Record }; + +function parseExactObject(text: string): ParseResult { + const trimmed = text.trim(); + if (trimmed.length === 0) { + return { kind: "incomplete" }; + } + if (trimmed.startsWith("{") === false) { + return { kind: "reject" }; + } + try { + const parsed: unknown = JSON.parse(trimmed); + if (!isPlainObject(parsed)) { + return { kind: "reject" }; + } + return { kind: "object", value: parsed }; + } catch { + return { kind: "incomplete" }; + } +} + +function salvageToolCall( + value: Record, + declaredNames: ReadonlySet, +): { name: string; args: Record } | null { + for (const key of Object.keys(value)) { + if ( + key !== "name" && + key !== "parameters" && + key !== "arguments" && + key !== "id" + ) { + return null; + } + } + const name = value["name"]; + if (typeof name !== "string" || declaredNames.has(name) === false) { + return null; + } + const hasParameters = Object.hasOwn(value, "parameters"); + const hasArguments = Object.hasOwn(value, "arguments"); + if (hasParameters === hasArguments) { + return null; + } + const raw = hasParameters ? value["parameters"] : value["arguments"]; + if (!isPlainObject(raw)) { + return null; + } + const id = value["id"]; + if (id !== undefined && typeof id !== "string") { + return null; + } + return { name, args: raw }; +} + +function lastHeldText( + held: readonly InferenceEvent[], +): InferenceEvent | undefined { + for (let i = held.length - 1; i >= 0; i--) { + const event = held[i]; + if (event?.type === "inference.text.delta") { + return event; + } + } + return undefined; +} + +function emitToolCallEvents( + salvaged: { name: string; args: Record }, + template: InferenceEvent | undefined, +): InferenceEvent[] { + const seq = template?.seq ?? 0; + const index = + template?.type === "inference.text.delta" && + template.data.index !== undefined + ? template.data.index + : 0; + return [ + { + type: "inference.tool_call.start", + seq, + data: { + callId: INLINE_TOOL_CALL_ID, + name: salvaged.name, + partial: EMPTY_PARTIAL, + index, + }, + }, + { + type: "inference.tool_call.delta", + seq, + data: { + // Same id as start. The harness keys open tool calls by start's + // callId and resolves a delta as `indexToCallId.get(callId) ?? + // callId`; a matching id attaches arguments. OpenAI's + // `String(index)` placeholder is only for native continuation + // deltas that omit the real id — this salvage mints both events. + callId: INLINE_TOOL_CALL_ID, + argumentFragment: JSON.stringify(salvaged.args), + partial: EMPTY_PARTIAL, + index, + }, + }, + ]; +} + +function matchingSalvage( + state: InlineToolJsonState, +): { name: string; args: Record } | null { + const parsed = parseExactObject(state.acc); + if (parsed.kind !== "object") { + return null; + } + return salvageToolCall(parsed.value, state.declaredNames); +} + +function releaseHeldAsText( + output: InferenceEvent[], + state: InlineToolJsonState, +): void { + output.push(...state.held); + state.held = []; + state.verdict = "text"; +} + +function flushPending( + output: InferenceEvent[], + state: InlineToolJsonState, +): void { + if (state.verdict !== "pending") { + return; + } + if (state.held.length === 0 && state.acc === "") { + return; + } + const salvaged = matchingSalvage(state); + if (salvaged !== null) { + output.push(...emitToolCallEvents(salvaged, lastHeldText(state.held))); + state.held = []; + state.verdict = "tool"; + return; + } + releaseHeldAsText(output, state); +} + +function inspectHeldText( + output: InferenceEvent[], + state: InlineToolJsonState, +): void { + const parsed = parseExactObject(state.acc); + if (parsed.kind === "incomplete") { + return; + } + if (parsed.kind === "reject") { + releaseHeldAsText(output, state); + return; + } + if (salvageToolCall(parsed.value, state.declaredNames) === null) { + releaseHeldAsText(output, state); + } +} + +export function responseChunkIsTerminal(sseData: string): boolean { + try { + const parsed: unknown = JSON.parse(sseData); + if (!isPlainObject(parsed)) { + return false; + } + if (parsed["usage"] != null) { + return true; + } + const choices = parsed["choices"]; + if (!Array.isArray(choices) || choices.length === 0) { + return false; + } + const first: unknown = choices[0]; + if (!isPlainObject(first)) { + return false; + } + const reason = first["finish_reason"]; + return typeof reason === "string" && reason.length > 0; + } catch { + return false; + } +} + +/** + * Rewrites one `parseResponse`/`parseJSONResponse` result so a content + * stream that is exactly a declared-tool JSON object becomes tool-call + * events instead of text. `state` is mutated in place so a caller threads + * the same instance across every chunk of one response. Pass `flush` on + * the terminal chunk (finish_reason / complete JSON body) so a matching + * object is committed only once the stream cannot grow trailing prose. + */ +export function reclassifyInlineToolJsonEvents( + events: readonly InferenceEvent[], + state: InlineToolJsonState, + opts?: { flush?: boolean }, +): InferenceEvent[] { + if (state.declaredNames.size === 0) { + return [...events]; + } + + const output: InferenceEvent[] = []; + + for (const event of events) { + if (state.verdict === "text") { + output.push(event); + continue; + } + if (state.verdict === "tool") { + if (event.type !== "inference.text.delta") { + output.push(event); + } + continue; + } + if (event.type === "inference.text.delta") { + state.acc += event.data.token; + state.held.push(event); + inspectHeldText(output, state); + continue; + } + if (event.type === "inference.usage") { + flushPending(output, state); + output.push(event); + continue; + } + if ( + event.type === "inference.tool_call.start" || + event.type === "inference.tool_call.delta" || + event.type === "inference.tool_call.end" + ) { + releaseHeldAsText(output, state); + output.push(event); + continue; + } + output.push(event); + } + + if (opts?.flush === true) { + flushPending(output, state); + } + + return output; +} diff --git a/workflows/assistant/src/index.ts b/workflows/assistant/src/index.ts index 3d6a6dd52..c5190ad44 100644 --- a/workflows/assistant/src/index.ts +++ b/workflows/assistant/src/index.ts @@ -213,14 +213,18 @@ export const ASSISTANT_SYSTEM_PROMPT = "\n" + "## Tools\n" + "Each tool's own description says how it works; what spans them: " + + "invoke tools only through tool calls, never by writing a JSON object " + + "with a tool name into your reply. " + "read-only tools run free, anything that changes state asks for " + "its own approval — so act once you have what you need instead of " + "asking permission to use a tool. Use the team's firm memory " + "(memory_search, memory_add, memory_list) to recall facts and " + "decisions from earlier conversations and to record ones worth " + - "keeping — never fabricate a recollection when a search comes back " + - "empty, and if memory isn't set up on this deployment, proceed " + - "without mentioning it. Any MCP server connected under Plugins is " + + "keeping — never memory_search a bare greeting, and use memory only " + + "when you actually need a fact from earlier; never fabricate a " + + "recollection when a search comes back empty, and if memory isn't " + + "set up on this deployment, proceed without mentioning it. Any MCP " + + "server connected under Plugins is " + "reachable with mcp_list_servers, mcp_list_tools, mcp_read, and " + "mcp_call — discover once with mcp_list_tools (pattern search when " + "unsure which server has the tool you want); use mcp_read for " + diff --git a/workflows/assistant/test/definition.test.ts b/workflows/assistant/test/definition.test.ts index 5be5b32ab..6a728793c 100644 --- a/workflows/assistant/test/definition.test.ts +++ b/workflows/assistant/test/definition.test.ts @@ -150,6 +150,19 @@ test("the prompt tells Myra to discover an MCP server's tools before calling one expect(ASSISTANT_SYSTEM_PROMPT).toContain("never guess a tool name"); }); +test("the prompt never writes tool JSON as reply text and does not memory_search a bare greeting", () => { + expect(ASSISTANT_SYSTEM_PROMPT).toContain("only through tool calls"); + expect(ASSISTANT_SYSTEM_PROMPT).toContain( + "never by writing a JSON object with a tool name into your reply", + ); + expect(ASSISTANT_SYSTEM_PROMPT).toContain( + "never memory_search a bare greeting", + ); + expect(ASSISTANT_SYSTEM_PROMPT).toContain( + "use memory only when you actually need a fact from earlier", + ); +}); + test("the workflow pins manus-tools and does not require a Manus credential binding", () => { const definition = buildAssistantWorkflow(INPUT); expect(ASSISTANT_TOOL_PACKAGE_PINS.map((pin) => pin.name)).toContain( From f71a76c5a750558875592748ba26ffde07795018 Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Sat, 29 Aug 2026 21:07:36 -0700 Subject: [PATCH 2/2] Update docs: Ollama inline tool calls --- ARCHITECTURE.md | 10 ++++++++++ IMPLEMENTATION.md | 19 +++++++++++++++++++ PRODUCT.md | 5 ++++- 3 files changed, 33 insertions(+), 1 deletion(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index c6f487b6e..fc6a8b635 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -154,6 +154,16 @@ one path, deltas to pixels: synthetic in-progress message alongside the persisted timeline, until the real message lands and replaces it. +**Provider tool-call quirks.** Some OpenAI-compatible local models emit +a declared tool call as a JSON object in assistant text instead of a +native tool-call field. Workbench's Ollama adapter wraps the platform +OpenAI adapter (composition, not a fork) and reclassifies that exact +object into structured tool-call events when the tool name was +declared for the request. Unknown names, extra keys, arrays, mixed +prose, and fenced JSON stay text so a legitimate JSON or code answer +is not stolen. The reactor and timeline still only see the usual text +versus tool-call events. + ## Workbench Definition and hostless onboarding A picker "template" is a shipped **Workbench Definition**, not a second diff --git a/IMPLEMENTATION.md b/IMPLEMENTATION.md index 624321fb2..719e627b5 100644 --- a/IMPLEMENTATION.md +++ b/IMPLEMENTATION.md @@ -259,6 +259,25 @@ non-empty owned list restricts candidates to those names; an empty owned list keeps inherited discovery. Other providers still pin `CATALOG_SEEDS[provider].models[0]`. +Ollama's OpenAI-compatible endpoint can also emit a declared tool call +as a JSON object in `content` instead of native `tool_calls`. +`@intx/inference`'s OpenAI parser only reads `delta.tool_calls`, so +that JSON would otherwise land on the timeline as assistant text and +the tool would never run. `@corbits/ollama-adapter` +(`reclassifyInlineToolJsonEvents` in +`packages/ollama-adapter/src/inline-tool-json.ts`) rewrites a content +stream that is exactly that object into `inference.tool_call.*` +events, gated on the tools declared for the request. Salvage requires +a plain object whose keys are only `name` plus exactly one of +`parameters` or `arguments` (optional `id`), and a `name` in the +declared set. Unknown names, no-tools requests, extra keys, arrays, +mixed prose, fenced JSON, and incomplete objects stay text. + +The assistant definition (`workflows/assistant`) also tells Myra to +invoke tools only through tool calls — never by writing a JSON object +with a tool name into the reply — and not to `memory_search` a bare +greeting. + ## Related docs - [README.md](README.md) — quickstart, local setup, repo layout, e2e detail diff --git a/PRODUCT.md b/PRODUCT.md index 78dfe1a99..eadbdc63a 100644 --- a/PRODUCT.md +++ b/PRODUCT.md @@ -74,7 +74,10 @@ actually pulled, so Myra's first DM message is a real agent turn — even on a machine that only has models like llama3.2 or qwen3. An inherited catalog seed the instance never pulled is not the default when the tenant already owns a pulled completion model. Embedding models never -become that default. +become that default. A local model that writes a tool it meant to +invoke as JSON in the reply still becomes a human turn: the person +sees Myra's words (and ordinary tool activity), not that JSON as the +message. Create stays on `/new` (`apps/web/src/pages/new-workbench-picker.tsx`): a prompt box is the primary act: typing a goal and submitting mints an