From a8c14dd7cf7042fb3a470e319c8d9f014f5989a7 Mon Sep 17 00:00:00 2001 From: francisco-orkes Date: Fri, 28 Aug 2026 18:31:05 -0700 Subject: [PATCH 1/3] fix(agents): name tool calls from the payload, and populate AgentResult.Events `ToolCalls` named every tool after its Conductor task type. That is correct only for a worker tool, because Conductor sets an executed SIMPLE task's `taskType` to the task's own name; every other kind carries its system task type there, so an HTTP tool was named "HTTP", an MCP tool "CALL_MCP_TOOL", an agent used as a tool "SUB_WORKFLOW". Resolve the name from `inputData._agent_tool_name`, then `inputData.method`, then `taskDefName`. Detection required a `call_` reference-name prefix. The server seeds the reference from the provider's tool-call id, so that prefix is OpenAI's format and an Anthropic-backed agent recorded no tool calls at all. Identify a tool task by task type instead, allowlisted off the server's `ToolCompiler.TYPE_MAP`. Two cases need corroborating evidence that the LLM dispatched the task, and take a dispatch marker (`_agent_tool_name` or `_agent_state`) as well: the worker kind, whose task type is the tool's own name and so cannot be allowlisted, and `SUB_WORKFLOW`/`HUMAN`, which the agent layer emits for sub-agents, strategy workflows, routers and plan approval steps far more often than for a tool. Without that second requirement every multi-agent handoff would be reported as a tool call that never happened. `AgentResult.Events` was never assigned, so enumerating it threw. Build it from the same task walk: a `ToolCall`/`ToolResult` pair per tool call, closed by a terminal `Done`, or `Error` for a run that did not complete. It is a reconstruction from the finished execution's tasks and is narrower than the live stream, which is documented rather than implied. Also read a tool's result from the whole task output when there is no `result` key, since an HTTP tool answers under `response`, and include `GENERATE_PDF`, which `TYPE_MAP` omits although the server compiles a `generate_pdf` tool to it. The existing tool-call test passed for the right reason but on a fixture that omitted `method` and `_agent_tool_name` and carried a reference name without the fork index and loop suffix real payloads have, so it exercised neither name resolution nor detection. Its fixture now has the shape a real payload does. --- Conductor.AI.Tests/AgentStatusMappingTests.cs | 40 +- .../AgentToolCallExtractionTests.cs | 372 ++++++++++++++++++ Conductor.AI.Tests/StubAgentServer.cs | 65 +++ Conductor.AI/Result.cs | 234 +++++++++-- docs/agents/concepts/streaming-hitl.md | 9 + docs/agents/reference/api.md | 4 +- 6 files changed, 659 insertions(+), 65 deletions(-) create mode 100644 Conductor.AI.Tests/AgentToolCallExtractionTests.cs create mode 100644 Conductor.AI.Tests/StubAgentServer.cs diff --git a/Conductor.AI.Tests/AgentStatusMappingTests.cs b/Conductor.AI.Tests/AgentStatusMappingTests.cs index bfd8fc55..4e346bd5 100644 --- a/Conductor.AI.Tests/AgentStatusMappingTests.cs +++ b/Conductor.AI.Tests/AgentStatusMappingTests.cs @@ -26,30 +26,8 @@ namespace Conductor.AI.Tests; public sealed class AgentStatusMappingTests { - private sealed class StubHandler : HttpMessageHandler - { - private readonly Func _respond; - public StubHandler(Func respond) => _respond = respond; - - protected override Task SendAsync(HttpRequestMessage request, CancellationToken ct) - { - var (status, body) = _respond(request); - return Task.FromResult(new HttpResponseMessage(status) - { - Content = new StringContent(body, Encoding.UTF8, "application/json"), - }); - } - } - private static Configuration BuildConfig(Func respond) - { - var configuration = new Configuration { BasePath = "http://server/api" }; - configuration.ApiClient.RestClient = new RestClient(new RestClientOptions("http://server/api") - { - ConfigureMessageHandler = _ => new StubHandler(respond), - }); - return configuration; - } + => StubAgentServer.Configure(respond); private static (HttpStatusCode, string) RouteStatusAndExecution( HttpRequestMessage request, string statusBody, string executionBody = "{}") @@ -57,12 +35,7 @@ private static (HttpStatusCode, string) RouteStatusAndExecution( private static (HttpStatusCode, string) Route( HttpRequestMessage request, string statusBody, string executionBody, string workflowBody) - { - var path = request.RequestUri!.AbsolutePath; - if (path.EndsWith("/status")) return (HttpStatusCode.OK, statusBody); - if (path.Contains("/execution/")) return (HttpStatusCode.OK, executionBody); - return (HttpStatusCode.OK, workflowBody); - } + => StubAgentServer.Route(request, statusBody, executionBody, workflowBody); // ── AgentRuntime.GetStatusAsync ────────────────────────────────────── @@ -171,11 +144,12 @@ public async Task WaitAsync_ExtractsToolCallsFromWorkflowTasks() executionBody: "{}", workflowBody: """ {"tasks":[ - {"taskType":"echo","referenceTaskName":"call_echo_1", - "inputData":{"query":"hello","_internal":"drop me","ctx":"drop me too"}, + {"taskType":"echo","taskDefName":"echo","referenceTaskName":"call_ceOwlp7lQ_0__1", + "inputData":{"query":"hello","method":"echo","_agent_tool_name":"echo", + "_agent_state":{},"_internal":"drop me","ctx":"drop me too"}, "outputData":{"result":"echoed: hello"}}, - {"taskType":"LLM_CHAT_COMPLETE","referenceTaskName":"llm_1", - "inputData":{},"outputData":{"promptTokens":10}} + {"taskType":"LLM_CHAT_COMPLETE","taskDefName":"llm_chat_complete", + "referenceTaskName":"llm_1","inputData":{},"outputData":{"promptTokens":10}} ]} """)); diff --git a/Conductor.AI.Tests/AgentToolCallExtractionTests.cs b/Conductor.AI.Tests/AgentToolCallExtractionTests.cs new file mode 100644 index 00000000..c3f55318 --- /dev/null +++ b/Conductor.AI.Tests/AgentToolCallExtractionTests.cs @@ -0,0 +1,372 @@ +/* + * Copyright 2024 Conductor Authors. + *

+ * Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + *

+ * http://www.apache.org/licenses/LICENSE-2.0 + *

+ * Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on + * an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the + * specific language governing permissions and limitations under the License. + */ +// AgentResult.ToolCalls used to name every tool after its Conductor task type — +// correct only for a worker tool, whose executed SIMPLE task carries the tool's +// own name there, and wrong for every other kind ("HTTP", "CALL_MCP_TOOL", +// "SUB_WORKFLOW", "HUMAN", "GENERATE_IMAGE"). Detection also keyed on a +// `call_` reference-name prefix, which is the OpenAI tool-call ID format, so an +// Anthropic-backed agent recorded no tool calls at all. +// +// Fixtures here carry the shape real payloads have: the fork index and loop +// suffix on the reference name, and the `_agent_tool_name` / `_agent_state` +// keys the server's dispatch script injects. + +using System.Linq; +using System.Net; +using System.Net.Http; +using System.Text.Json; +using System.Text; +using System.Threading.Tasks; +using Conductor.Client; +using RestSharp; +using Xunit; + +namespace Conductor.AI.Tests; + +public sealed class AgentToolCallExtractionTests +{ + ///

Drive WaitAsync to completion over a stubbed server and return the built result. + private static async Task WaitWithTasksAsync( + string workflowBody, string statusBody = DefaultStatus) + { + var configuration = StubAgentServer.Configure( + request => StubAgentServer.Route(request, statusBody, "{}", workflowBody)); + var handle = new AgentHandle("e1", new OrkesAgentClient(configuration)); + return await handle.WaitAsync(); + } + + private const string DefaultStatus = """ + {"executionId":"e1","status":"COMPLETED","isComplete":true,"isRunning":false, + "isWaiting":false,"output":{"result":"done"}} + """; + + private static Dictionary ArgsOf(Dictionary toolCall) + => Assert.IsType>(toolCall["args"]); + + // ── Tool name comes from the payload, never from the task type ─────── + + [Fact] + public async Task WorkerTool_NameFromAgentToolName() + { + // A worker tool's executed SIMPLE task carries the tool name in taskType + // too, so this is the one kind the old taskType read got right. It is + // here to pin that the marker-based resolution agrees with it. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"get_weather","taskDefName":"get_weather", + "referenceTaskName":"call_PMnNIdOPvm9EQ8e6tn2kbxPY_0__1", + "inputData":{"_agent_tool_name":"get_weather","_agent_state":{}, + "method":"get_weather","city":"San Francisco"}, + "outputData":{"result":"Sunny, 72F"}} + ]} + """); + + var toolCall = Assert.Single(result.ToolCalls!); + Assert.Equal("get_weather", toolCall["name"]); + Assert.Equal(["city"], ArgsOf(toolCall).Keys); + Assert.Equal("Sunny, 72F", toolCall["result"]!.ToString()); + } + + [Theory] + // Every non-worker kind the server's ToolCompiler.TYPE_MAP can produce: the + // task type is the system type, and the tool's real name is in inputData. + [InlineData("HTTP", "lookup_order")] + [InlineData("CALL_MCP_TOOL", "search_docs")] + [InlineData("SUB_WORKFLOW", "billing_agent")] + [InlineData("HUMAN", "escalate")] + [InlineData("GENERATE_IMAGE", "draw_chart")] + [InlineData("GENERATE_AUDIO", "read_aloud")] + [InlineData("GENERATE_VIDEO", "animate")] + [InlineData("GENERATE_PDF", "make_report")] + [InlineData("LLM_INDEX_TEXT", "index_docs")] + [InlineData("LLM_SEARCH_INDEX", "search_index")] + [InlineData("PULL_WORKFLOW_MESSAGES", "read_queue")] + public async Task SystemTaskTool_NameFromAgentToolName_NotTaskType(string taskType, string toolName) + { + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"TYPE","taskDefName":"NAME","referenceTaskName":"NAME_0__1", + "inputData":{"_agent_tool_name":"NAME","query":"x"}, + "outputData":{"result":"ok"}} + ]} + """.Replace("TYPE", taskType).Replace("NAME", toolName)); + + var toolCall = Assert.Single(result.ToolCalls!); + Assert.Equal(toolName, toolCall["name"]); + Assert.NotEqual(taskType, toolCall["name"]); + } + + [Fact] + public async Task NameFallsBackToMethod_WhenAgentToolNameAbsent() + { + // The dynamic-tools dispatch script sets no `_agent_tool_name`, but an + // MCP task carries `method` — the tool name the LLM called. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"CALL_MCP_TOOL","taskDefName":"call_mcp_tool", + "referenceTaskName":"search_docs_0", + "inputData":{"mcpServer":"docs","method":"search_docs","arguments":{"q":"x"}}, + "outputData":{"result":["a"]}} + ]} + """); + + Assert.Equal("search_docs", Assert.Single(result.ToolCalls!)["name"]); + } + + [Fact] + public async Task NameFallsBackToTaskDefName_WhenNeitherMarkerPresent() + { + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"GENERATE_IMAGE","taskDefName":"generate_image", + "referenceTaskName":"generate_image_0", + "inputData":{"prompt":"a cat"}, + "outputData":{"result":"http://img"}} + ]} + """); + + Assert.Equal("generate_image", Assert.Single(result.ToolCalls!)["name"]); + } + + // ── Detection no longer keys on the provider's tool-call ID ────────── + + [Fact] + public async Task AnthropicReferenceName_StillDetected() + { + // Anthropic tool-call IDs start `toolu_`; the reference name is seeded + // from the provider's ID, so a `call_` prefix test found nothing here. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"get_weather","taskDefName":"get_weather", + "referenceTaskName":"toolu_01A09q90qw90lq917835lq9_0__1", + "inputData":{"_agent_tool_name":"get_weather","_agent_state":{},"city":"Paris"}, + "outputData":{"result":"Rainy"}} + ]} + """); + + Assert.Equal("get_weather", Assert.Single(result.ToolCalls!)["name"]); + } + + [Fact] + public async Task WorkerToolWithoutToolNameMarker_DetectedByAgentState() + { + // The dynamic-tools script injects `_agent_state` but not + // `_agent_tool_name`, and a worker tool's task type is the tool's own + // name, so it cannot be recognised from an allowlist of system types. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"echo","taskDefName":"echo","referenceTaskName":"toolu_xyz_0", + "inputData":{"_agent_state":{},"method":"echo","text":"hi"}, + "outputData":{"result":"hi"}} + ]} + """); + + Assert.Equal("echo", Assert.Single(result.ToolCalls!)["name"]); + } + + [Fact] + public async Task NonToolTasks_Excluded() + { + // The agent's own scaffolding, plus the INLINE task the dispatch script + // substitutes for a hallucinated tool name — none of these is a tool call. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"LLM_CHAT_COMPLETE","taskDefName":"llm_chat_complete", + "referenceTaskName":"llm_1","inputData":{},"outputData":{"promptTokens":10}}, + {"taskType":"FORK_JOIN_DYNAMIC","taskDefName":"fork","referenceTaskName":"fork_1", + "inputData":{},"outputData":{}}, + {"taskType":"JOIN","taskDefName":"join","referenceTaskName":"join_1", + "inputData":{},"outputData":{}}, + {"taskType":"INLINE","taskDefName":"made_up_tool","referenceTaskName":"made_up_tool", + "inputData":{"evaluatorType":"graaljs","expression":"...","errorMessage":"Unknown tool"}, + "outputData":{"result":"Unknown tool 'made_up_tool'.","is_error":true}}, + {"taskType":"SWITCH","taskDefName":"switch","referenceTaskName":"switch_1", + "inputData":{},"outputData":{}}, + {"taskType":"DO_WHILE","taskDefName":"loop","referenceTaskName":"loop_1", + "inputData":{},"outputData":{}}, + {"taskType":"SET_VARIABLE","taskDefName":"set_var","referenceTaskName":"set_var_1", + "inputData":{},"outputData":{}} + ]} + """); + + Assert.Null(result.ToolCalls); + } + + [Fact] + public async Task MultiAgentSetVariable_NotAToolCall() + { + // A multi-agent coordinator's SET_VARIABLE task carries `_agent_state` + // in its inputs, so the dispatch marker alone does not make a task a + // tool call — the worker case also needs the executed-SIMPLE signature + // of a task type equal to its own taskDefName. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"SET_VARIABLE","taskDefName":"triage_init", + "referenceTaskName":"triage_init", + "inputData":{"conversation":"hello","_agent_state":{"turn":1}}, + "outputData":{}} + ]} + """); + + Assert.Null(result.ToolCalls); + } + + [Fact] + public async Task WorkerTaskWithoutDispatchMarker_NotAToolCall() + { + // The framework-passthrough wrapper is a SIMPLE task named after a + // worker, but the LLM never dispatched it, so it carries no marker. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"claude_code","taskDefName":"claude_code", + "referenceTaskName":"_fw_task", + "inputData":{"prompt":"hi","session_id":"s1"}, + "outputData":{"result":"hello"}} + ]} + """); + + Assert.Null(result.ToolCalls); + } + + [Theory] + // The agent layer emits these types for its own structure: a sub-agent, a + // strategy workflow, a router and a plan execution are all SUB_WORKFLOW, and + // a plan's approval step is HUMAN. None is dispatched by the LLM, so none + // carries a dispatch marker, and none may be reported as a tool call. + [InlineData("SUB_WORKFLOW", "billing_strategy")] + [InlineData("SUB_WORKFLOW", "triage_router")] + [InlineData("HUMAN", "plan_approval")] + public async Task AgentStructureReusingAToolTaskType_NotAToolCall(string taskType, string taskDefName) + { + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"TYPE","taskDefName":"NAME","referenceTaskName":"0_NAME__1", + "inputData":{"prompt":"hello","media":[],"session_id":"s1", + "context":{"turn":1}}, + "outputData":{"result":"handled"}} + ]} + """.Replace("TYPE", taskType).Replace("NAME", taskDefName)); + + Assert.Null(result.ToolCalls); + Assert.Single(result.Events!); + } + + [Fact] + public async Task AgentAsTool_DetectedByDispatchMarker() + { + // The same SUB_WORKFLOW type is a genuine tool call when the dispatch + // script marked it as one. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"SUB_WORKFLOW","taskDefName":"billing_agent_wf", + "referenceTaskName":"toolu_01xyz_0__1", + "inputData":{"_agent_tool_name":"billing_agent","prompt":"refund order A-1", + "session_id":"s1"}, + "outputData":{"result":"refunded"}} + ]} + """); + + Assert.Equal("billing_agent", Assert.Single(result.ToolCalls!)["name"]); + } + + [Fact] + public async Task InternalInputKeys_StrippedFromArgs() + { + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"HTTP","taskDefName":"lookup","referenceTaskName":"lookup_0", + "inputData":{"_agent_tool_name":"lookup","_agent_state":{},"method":"lookup", + "ctx":"x","workerTag":"y","agentConfig":{},"order":"A-1"}, + "outputData":{"result":"shipped"}} + ]} + """); + + Assert.Equal(["order"], ArgsOf(Assert.Single(result.ToolCalls!)).Keys); + } + + // ── Events ────────────────────────────────────────────────────────── + + [Fact] + public async Task Events_ToolCallAndResultPerToolTask_ThenDone() + { + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"HTTP","taskDefName":"lookup_order","referenceTaskName":"lookup_order_0", + "inputData":{"_agent_tool_name":"lookup_order","order":"A-1"}, + "outputData":{"result":"shipped"}} + ]} + """); + + Assert.Collection( + result.Events!, + e => + { + Assert.Equal(EventType.ToolCall, e.Type); + Assert.Equal("lookup_order", e.ToolName); + Assert.Equal(["order"], e.Args!.Keys); + }, + e => + { + Assert.Equal(EventType.ToolResult, e.Type); + Assert.Equal("lookup_order", e.ToolName); + Assert.Equal("shipped", e.Result!.ToString()); + }, + e => + { + Assert.Equal(EventType.Done, e.Type); + Assert.Equal("e1", e.ExecutionId); + }); + } + + [Fact] + public async Task Events_NeverNull_SoEnumerationDoesNotThrow() + { + var result = await WaitWithTasksAsync("""{"tasks":[]}"""); + + Assert.NotNull(result.Events); + var terminal = Assert.Single(result.Events!); + Assert.Equal(EventType.Done, terminal.Type); + } + + [Fact] + public async Task Events_FailedRun_EndsWithErrorNotDone() + { + var result = await WaitWithTasksAsync("""{"tasks":[]}""", statusBody: """ + {"executionId":"e1","status":"FAILED","isComplete":true,"isRunning":false, + "isWaiting":false,"reasonForIncompletion":"tool worker never polled"} + """); + + var terminal = Assert.Single(result.Events!); + Assert.Equal(EventType.Error, terminal.Type); + Assert.Equal("tool worker never polled", terminal.Content); + } + + [Fact] + public async Task HttpToolResult_FallsBackToWholeOutput() + { + // An HTTP tool answers under `response`, not `result`, so keying only on + // `result` would report a tool call with no result at all. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"HTTP","taskDefName":"lookup_order","referenceTaskName":"lookup_order_0", + "inputData":{"_agent_tool_name":"lookup_order","order":"A-1"}, + "outputData":{"response":{"status":"shipped"},"statusCode":200}} + ]} + """); + + var toolResult = Assert.IsType(Assert.Single(result.ToolCalls!)["result"]); + Assert.Equal( + ["response", "statusCode"], + toolResult.EnumerateObject().Select(p => p.Name)); + } +} diff --git a/Conductor.AI.Tests/StubAgentServer.cs b/Conductor.AI.Tests/StubAgentServer.cs new file mode 100644 index 00000000..5cf5e4ad --- /dev/null +++ b/Conductor.AI.Tests/StubAgentServer.cs @@ -0,0 +1,65 @@ +/* + * Copyright 2024 Conductor Authors. + *

+ * Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + *

+ * http://www.apache.org/licenses/LICENSE-2.0 + *

+ * Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on + * an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the + * specific language governing permissions and limitations under the License. + */ +// A stubbed agent server for tests that drive AgentHandle against canned +// responses, so an execution's terminal state can be stated as a payload. + +using System.Net; +using System.Net.Http; +using System.Text; +using System.Threading.Tasks; +using Conductor.Client; +using RestSharp; + +namespace Conductor.AI.Tests; + +internal static class StubAgentServer +{ + private sealed class StubHandler : HttpMessageHandler + { + private readonly Func _respond; + public StubHandler(Func respond) => _respond = respond; + + protected override Task SendAsync(HttpRequestMessage request, CancellationToken ct) + { + var (status, body) = _respond(request); + return Task.FromResult(new HttpResponseMessage(status) + { + Content = new StringContent(body, Encoding.UTF8, "application/json"), + }); + } + } + + ///

A client configuration whose every request is answered by . + internal static Configuration Configure(Func respond) + { + var configuration = new Configuration { BasePath = "http://server/api" }; + configuration.ApiClient.RestClient = new RestClient(new RestClientOptions("http://server/api") + { + ConfigureMessageHandler = _ => new StubHandler(respond), + }); + return configuration; + } + + /// + /// Answer the three reads makes on reaching + /// a terminal state: the status, the execution record, and the workflow with its tasks. + /// + internal static (HttpStatusCode, string) Route( + HttpRequestMessage request, string statusBody, string executionBody, string workflowBody) + { + var path = request.RequestUri!.AbsolutePath; + if (path.EndsWith("/status")) return (HttpStatusCode.OK, statusBody); + if (path.Contains("/execution/")) return (HttpStatusCode.OK, executionBody); + return (HttpStatusCode.OK, workflowBody); + } +} diff --git a/Conductor.AI/Result.cs b/Conductor.AI/Result.cs index b69afa92..3a5c12bd 100644 --- a/Conductor.AI/Result.cs +++ b/Conductor.AI/Result.cs @@ -305,8 +305,9 @@ public async Task WaitAsync(CancellationToken cancellationToken = d { // Fetch full execution record for token usage and finish reason var execution = await _http.GetExecutionAsync(_executionId, cancellationToken); - // Java parity: walk the workflow's tasks once to aggregate tool calls - // from call_* tasks (enrichment read — null is fine, yields no tool calls). + // Walk the workflow's tasks to recover the tool calls and the + // events the run produced (enrichment read — null is fine, + // and yields a result carrying neither). var workflowWithTasks = await _http.GetWorkflowWithTasksAsync(_executionId, cancellationToken); return BuildResult(status!, s, execution, workflowWithTasks); } @@ -536,55 +537,226 @@ private static AgentResult BuildResult( _ => null, }; + var executionId = status["executionId"]?.GetValue() ?? ""; + var error = parsedStatus != Status.Completed + ? status["reasonForIncompletion"]?.GetValue() + : null; + + var (toolCalls, events) = ExtractToolActivity(workflowWithTasks, executionId); + events.Add(parsedStatus == Status.Completed + ? new AgentEvent { Type = EventType.Done, ExecutionId = executionId, Output = outputDict } + : new AgentEvent { Type = EventType.Error, ExecutionId = executionId, Content = error }); + return new AgentResult { - ExecutionId = status["executionId"]?.GetValue() ?? "", + ExecutionId = executionId, Status = parsedStatus, Output = outputDict, - Error = parsedStatus != Status.Completed ? status["reasonForIncompletion"]?.GetValue() : null, - ToolCalls = ExtractToolCalls(workflowWithTasks), + Error = error, + ToolCalls = toolCalls, TokenUsage = tokenUsage, FinishReason = finishReason, + Events = events, }; } + // ── Tool-call extraction ───────────────────────────────────────── + // + // A tool task is identified by its Conductor task type, allowlisted off the + // server's ToolCompiler.TYPE_MAP — for the types the agent layer also uses + // for its own structure, and for the worker kind, corroborated by a dispatch + // marker. It is never identified by its reference task name: the server + // seeds that from the provider's tool-call id (OpenAI's `call_...`, + // Anthropic's `toolu_...`, else a UUID) and appends the fork index and loop + // iteration, so a prefix test only ever matches one provider. + // + // The tool's real name is likewise never the task type. Conductor sets an + // executed SIMPLE task's type to the task's own name, which is the tool + // name for a worker tool and is why that one kind used to read correctly; + // every other kind carries its system task type there instead. + + /// + /// Task types only a tool compiles to, from the server's + /// ToolCompiler.TYPE_MAP — plus GENERATE_PDF, which that map + /// omits although the server compiles a generate_pdf tool to it. + /// SIMPLE is absent deliberately: a worker tool's executed task type + /// is the tool's own name, so that kind is recognised by the dispatch + /// markers below rather than by type. + /// + private static readonly HashSet ToolTaskTypes = new(StringComparer.Ordinal) + { + "HTTP", "CALL_MCP_TOOL", + "GENERATE_IMAGE", "GENERATE_AUDIO", "GENERATE_VIDEO", "GENERATE_PDF", + "LLM_INDEX_TEXT", "LLM_SEARCH_INDEX", "PULL_WORKFLOW_MESSAGES", + }; + + /// + /// Task types a tool compiles to that the agent layer also uses for its own + /// structure — SUB_WORKFLOW for a sub-agent, a strategy workflow, a + /// router and a plan execution; HUMAN for a plan's approval step. + /// The type alone therefore proves nothing, and a dispatch marker is + /// required as well. + /// + private static readonly HashSet AmbiguousToolTaskTypes = new(StringComparer.Ordinal) + { + "SUB_WORKFLOW", "HUMAN", + }; + + /// + /// Input keys the server's tool-dispatch script injects. + /// _agent_tool_name is set for every tool kind on the static + /// dispatch path; _agent_state is set for worker tools on both the + /// static and the dynamic-tools path. + /// + private const string AgentToolNameKey = "_agent_tool_name"; + private const string AgentStateKey = "_agent_state"; + + /// The dispatch method name — the tool name on the dynamic-tools path. + private const string MethodKey = "method"; + + /// + /// Whether a task is an LLM-dispatched tool call: an unambiguous tool task + /// type, or a type that needs corroborating evidence that the LLM dispatched + /// it. + /// + /// Two cases need that evidence. The worker case, because a worker tool's + /// task type is the tool's own name and so cannot be allowlisted; what marks + /// it is Conductor setting an executed SIMPLE task's type to the task's own + /// name, which is also its taskDefName — a signature no system task + /// shares. And the ambiguous types above, which the agent layer emits for + /// its own structure far more often than for a tool. + /// + /// Neither half of the test suffices alone: a multi-agent + /// SET_VARIABLE task carries _agent_state without being a tool + /// call, and a multi-agent handoff is a SUB_WORKFLOW without being + /// one either. + /// + /// Accepted cost: the dynamic-tools dispatch path sets no marker on a + /// non-worker tool, so an agent-as-tool or human tool dispatched that way is + /// missed. That is the right way to be wrong — the alternative reports a + /// fabricated tool call for every handoff in every multi-agent run, and a + /// tool call that did not happen is worse than one that is absent. + /// + private static bool IsToolTask(JsonNode task) + { + var taskType = task["taskType"]?.GetValue(); + if (taskType is null) return false; + if (ToolTaskTypes.Contains(taskType)) return true; + + var needsMarker = AmbiguousToolTaskTypes.Contains(taskType) + || taskType == task["taskDefName"]?.GetValue(); + return needsMarker + && task["inputData"] is JsonObject inputData + && (inputData.ContainsKey(AgentToolNameKey) || inputData.ContainsKey(AgentStateKey)); + } + /// - /// Java parity (AgentHandle.extractFromTasks): walk the workflow's tasks - /// and collect one entry per LLM-dispatched tool call — tasks whose - /// referenceTaskName starts with call_ — capturing the tool name, - /// its input args (internal runtime fields stripped), and its result. + /// Resolve a tool task's tool name: inputData._agent_tool_name, then + /// inputData.method (the dynamic-tools dispatch path sets no + /// _agent_tool_name), then taskDefName. /// - private static List>? ExtractToolCalls(JsonNode? workflowWithTasks) + private static string ResolveToolName(JsonNode task) { - if (workflowWithTasks?["tasks"] is not JsonArray tasks) return null; + if (task["inputData"] is JsonObject inputData) + { + if (StringValue(inputData, AgentToolNameKey) is { } toolName) return toolName; + if (StringValue(inputData, MethodKey) is { } method) return method; + } + return task["taskDefName"]?.GetValue() ?? ""; + } + + private static string? StringValue(JsonObject obj, string key) + => obj.TryGetPropertyValue(key, out var node) && node is JsonValue value + && value.TryGetValue(out var s) && !string.IsNullOrEmpty(s) + ? s + : null; + /// The tool's arguments — the task's input with the runtime's own keys stripped. + private static Dictionary ToolArgs(JsonObject inputData) + { + var cleaned = new Dictionary(); + foreach (var kv in inputData) + { + var k = kv.Key; + if (k.StartsWith('_') || k == MethodKey || k is "evaluatorType" or "expression" or "ctx" + or "workerTag" or "agentConfig") + continue; + cleaned[k] = JsonSerializer.Deserialize(kv.Value?.ToJsonString() ?? "null", ConductorAgentJson.Options)!; + } + return cleaned; + } + + /// + /// The tool's result: the task's result output, or its whole output + /// when there is no such key — an HTTP tool answers under response, + /// so keying only on result would report no result at all for it. + /// Matches the fallback the server's AgentEventListener applies. + /// + private static object? ToolResult(JsonObject outputData) + { + var node = outputData.TryGetPropertyValue("result", out var resultNode) && resultNode is not null + ? resultNode + : outputData; + return JsonSerializer.Deserialize(node.ToJsonString(), ConductorAgentJson.Options); + } + + /// + /// Walk the workflow's tasks once and recover, per LLM-dispatched tool call, + /// both the entry (name, arguments, + /// result) and the / + /// event pair — two views of the same + /// call, so they are built together and cannot drift apart. + /// + /// The returned event list carries the tool events only; the caller appends + /// the terminal event. It is never null, so enumerating + /// never throws. Tool calls stay null when + /// there were none, as they always have. + /// + /// This is a reconstruction from completed tasks, not the stream + /// delivers live: the server emits events for + /// thinking steps, failed tasks, handoffs and guardrails, none of which + /// survives into the terminal record. Use when the + /// events themselves are the point. + /// + private static (List>? ToolCalls, List Events) ExtractToolActivity( + JsonNode? workflowWithTasks, string executionId) + { List>? toolCalls = null; + var events = new List(); + + if (workflowWithTasks?["tasks"] is not JsonArray tasks) return (toolCalls, events); + foreach (var task in tasks) { - var refName = task?["referenceTaskName"]?.GetValue(); - var outputData = task?["outputData"] as JsonObject; - if (refName is null || !refName.StartsWith("call_", StringComparison.Ordinal) || outputData is null) + if (task is null || task["outputData"] is not JsonObject outputData || !IsToolTask(task)) continue; - var tc = new Dictionary { ["name"] = task!["taskType"]?.GetValue() ?? "" }; - if (task["inputData"] is JsonObject inputData) - { - var cleaned = new Dictionary(); - foreach (var kv in inputData) - { - var k = kv.Key; - if (k.StartsWith('_') || k is "method" or "evaluatorType" or "expression" or "ctx" - or "workerTag" or "agentConfig") - continue; - cleaned[k] = JsonSerializer.Deserialize(kv.Value?.ToJsonString() ?? "null", ConductorAgentJson.Options)!; - } - tc["args"] = cleaned; - } - if (outputData.TryGetPropertyValue("result", out var resultNode) && resultNode is not null) - tc["result"] = JsonSerializer.Deserialize(resultNode.ToJsonString(), ConductorAgentJson.Options)!; + var name = ResolveToolName(task); + var args = task["inputData"] is JsonObject inputData ? ToolArgs(inputData) : null; + var result = ToolResult(outputData); + var tc = new Dictionary { ["name"] = name }; + if (args is not null) tc["args"] = args; + if (result is not null) tc["result"] = result; (toolCalls ??= new List>()).Add(tc); + + events.Add(new AgentEvent + { + Type = EventType.ToolCall, + ExecutionId = executionId, + ToolName = name, + Args = args, + Timestamp = task["startTime"]?.GetValue(), + }); + events.Add(new AgentEvent + { + Type = EventType.ToolResult, + ExecutionId = executionId, + ToolName = name, + Result = result, + Timestamp = task["endTime"]?.GetValue(), + }); } - return toolCalls; + return (toolCalls, events); } } diff --git a/docs/agents/concepts/streaming-hitl.md b/docs/agents/concepts/streaming-hitl.md index bf0f1b5a..c2c231eb 100644 --- a/docs/agents/concepts/streaming-hitl.md +++ b/docs/agents/concepts/streaming-hitl.md @@ -29,6 +29,15 @@ await foreach (var ev in runtime.StreamAsync(agent, "Write a haiku about C#.")) Event types: `Thinking`, `ToolCall`, `ToolResult`, `GuardrailPass`, `GuardrailFail`, `Waiting`, `Handoff`, `Message`, `Error`, `Done`. +### Events on a waited result + +`AgentResult.Events` carries the run's tool activity — a `ToolCall`/`ToolResult` pair +per tool call, closed by a terminal `Done`, or `Error` for a run that did not +complete. It is reconstructed from the finished execution's tasks, so it is never +null but is narrower than the live stream: the server also emits `Thinking`, +`Handoff`, guardrail and per-failed-task events, and none of those survives into the +terminal record. Stream the run when the events themselves are the point. + Streaming attempts SSE first and falls back to status-polling. Disable SSE entirely with `CONDUCTOR_AGENT_STREAMING_ENABLED=false` — see [deploy-serve-run.md](deploy-serve-run.md#worker-tuning-and-agentconfig). If the diff --git a/docs/agents/reference/api.md b/docs/agents/reference/api.md index f68439ce..bb749b2c 100644 --- a/docs/agents/reference/api.md +++ b/docs/agents/reference/api.md @@ -211,7 +211,9 @@ Positions map to server task names: `before_agent`, `after_agent`, `before_model **`AgentResult`** (record): `ExecutionId`, `CorrelationId`, `Output` (`Dictionary?`; final text usually `Output["result"]`), `Messages`, -`ToolCalls`, `Status`, `FinishReason`, `Error`, `TokenUsage`, `Metadata`, `Events`, +`ToolCalls`, `Status`, `FinishReason`, `Error`, `TokenUsage`, `Metadata`, `Events` +(tool activity plus a terminal event, reconstructed from the execution's tasks — +see [streaming-hitl.md](../concepts/streaming-hitl.md#events-on-a-waited-result)), `SubResults`. Convenience: `IsSuccess`, `IsFailed`, `IsRejected`, `PrintResult()`. **`AgentHandle`** — `ExecutionId`, `RunId`, `WaitAsync(ct)`, `StreamAsync(ct)`, From 3571c1f5c1623bce26e1089c6711b26064bbde6a Mon Sep 17 00:00:00 2001 From: francisco-orkes Date: Fri, 28 Aug 2026 19:26:18 -0700 Subject: [PATCH 2/3] docs(agents): trim tool-extraction comments to the file's convention The comment layer added with the tool-call fix explained how the code works where the code already shows it, and ran well past what this file does elsewhere: its 25 existing doc blocks average 2.5 lines, 17 of them are one-liners, and its section banners carry no prose at all. Docs now state what each member is, and why it exists only where that is not evident from the code: GENERATE_PDF is absent from the server's TYPE_MAP, the dispatch marker is what keeps the agent's own structure out of ToolCalls, and an HTTP tool answers under `response`. The fallback order, the return shape and the per-kind task type lists are left to the code. Test comments come down to the 5-6% of lines the sibling suites carry; the fixtures already show the payload shapes the prose was restating. Nothing executable changed. --- .../AgentToolCallExtractionTests.cs | 56 ++++------ Conductor.AI/Result.cs | 101 ++++-------------- 2 files changed, 38 insertions(+), 119 deletions(-) diff --git a/Conductor.AI.Tests/AgentToolCallExtractionTests.cs b/Conductor.AI.Tests/AgentToolCallExtractionTests.cs index c3f55318..c27f94fb 100644 --- a/Conductor.AI.Tests/AgentToolCallExtractionTests.cs +++ b/Conductor.AI.Tests/AgentToolCallExtractionTests.cs @@ -10,16 +10,10 @@ * an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the * specific language governing permissions and limitations under the License. */ -// AgentResult.ToolCalls used to name every tool after its Conductor task type — -// correct only for a worker tool, whose executed SIMPLE task carries the tool's -// own name there, and wrong for every other kind ("HTTP", "CALL_MCP_TOOL", -// "SUB_WORKFLOW", "HUMAN", "GENERATE_IMAGE"). Detection also keyed on a -// `call_` reference-name prefix, which is the OpenAI tool-call ID format, so an -// Anthropic-backed agent recorded no tool calls at all. -// -// Fixtures here carry the shape real payloads have: the fork index and loop -// suffix on the reference name, and the `_agent_tool_name` / `_agent_state` -// keys the server's dispatch script injects. +// AgentResult.ToolCalls named every tool after its Conductor task type, which is +// the tool's own name only for a worker tool, and detection keyed on a `call_` +// reference-name prefix, which is OpenAI's tool-call ID format. Fixtures carry the +// shape real payloads have: fork index, loop suffix and the dispatch markers. using System.Linq; using System.Net; @@ -58,9 +52,7 @@ private static Dictionary ArgsOf(Dictionary tool [Fact] public async Task WorkerTool_NameFromAgentToolName() { - // A worker tool's executed SIMPLE task carries the tool name in taskType - // too, so this is the one kind the old taskType read got right. It is - // here to pin that the marker-based resolution agrees with it. + // The one kind the old taskType read got right; pinned so it stays right. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"get_weather","taskDefName":"get_weather", @@ -78,8 +70,7 @@ public async Task WorkerTool_NameFromAgentToolName() } [Theory] - // Every non-worker kind the server's ToolCompiler.TYPE_MAP can produce: the - // task type is the system type, and the tool's real name is in inputData. + // Every non-worker kind in ToolCompiler.TYPE_MAP: task type is the system type. [InlineData("HTTP", "lookup_order")] [InlineData("CALL_MCP_TOOL", "search_docs")] [InlineData("SUB_WORKFLOW", "billing_agent")] @@ -109,8 +100,7 @@ public async Task SystemTaskTool_NameFromAgentToolName_NotTaskType(string taskTy [Fact] public async Task NameFallsBackToMethod_WhenAgentToolNameAbsent() { - // The dynamic-tools dispatch script sets no `_agent_tool_name`, but an - // MCP task carries `method` — the tool name the LLM called. + // The dynamic-tools script sets no `_agent_tool_name`; MCP carries `method`. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"CALL_MCP_TOOL","taskDefName":"call_mcp_tool", @@ -143,8 +133,7 @@ public async Task NameFallsBackToTaskDefName_WhenNeitherMarkerPresent() [Fact] public async Task AnthropicReferenceName_StillDetected() { - // Anthropic tool-call IDs start `toolu_`; the reference name is seeded - // from the provider's ID, so a `call_` prefix test found nothing here. + // Anthropic IDs start `toolu_`, so the old `call_` prefix test found nothing. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"get_weather","taskDefName":"get_weather", @@ -160,9 +149,8 @@ public async Task AnthropicReferenceName_StillDetected() [Fact] public async Task WorkerToolWithoutToolNameMarker_DetectedByAgentState() { - // The dynamic-tools script injects `_agent_state` but not - // `_agent_tool_name`, and a worker tool's task type is the tool's own - // name, so it cannot be recognised from an allowlist of system types. + // Dynamic dispatch injects `_agent_state` only, and a worker's task type is + // its own name, so no allowlist of system types can catch it. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"echo","taskDefName":"echo","referenceTaskName":"toolu_xyz_0", @@ -177,8 +165,7 @@ public async Task WorkerToolWithoutToolNameMarker_DetectedByAgentState() [Fact] public async Task NonToolTasks_Excluded() { - // The agent's own scaffolding, plus the INLINE task the dispatch script - // substitutes for a hallucinated tool name — none of these is a tool call. + // Agent scaffolding, plus the INLINE task substituted for a hallucinated name. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"LLM_CHAT_COMPLETE","taskDefName":"llm_chat_complete", @@ -205,10 +192,8 @@ public async Task NonToolTasks_Excluded() [Fact] public async Task MultiAgentSetVariable_NotAToolCall() { - // A multi-agent coordinator's SET_VARIABLE task carries `_agent_state` - // in its inputs, so the dispatch marker alone does not make a task a - // tool call — the worker case also needs the executed-SIMPLE signature - // of a task type equal to its own taskDefName. + // A coordinator's SET_VARIABLE carries `_agent_state`, so the marker alone is + // not enough: the worker case also needs taskType to equal taskDefName. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"SET_VARIABLE","taskDefName":"triage_init", @@ -224,8 +209,7 @@ public async Task MultiAgentSetVariable_NotAToolCall() [Fact] public async Task WorkerTaskWithoutDispatchMarker_NotAToolCall() { - // The framework-passthrough wrapper is a SIMPLE task named after a - // worker, but the LLM never dispatched it, so it carries no marker. + // A framework passthrough is a SIMPLE task the LLM never dispatched. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"claude_code","taskDefName":"claude_code", @@ -239,10 +223,8 @@ public async Task WorkerTaskWithoutDispatchMarker_NotAToolCall() } [Theory] - // The agent layer emits these types for its own structure: a sub-agent, a - // strategy workflow, a router and a plan execution are all SUB_WORKFLOW, and - // a plan's approval step is HUMAN. None is dispatched by the LLM, so none - // carries a dispatch marker, and none may be reported as a tool call. + // The agent layer emits these for its own structure (sub-agent, strategy, router, + // plan approval). None is LLM-dispatched, so none carries a marker. [InlineData("SUB_WORKFLOW", "billing_strategy")] [InlineData("SUB_WORKFLOW", "triage_router")] [InlineData("HUMAN", "plan_approval")] @@ -264,8 +246,7 @@ public async Task AgentStructureReusingAToolTaskType_NotAToolCall(string taskTyp [Fact] public async Task AgentAsTool_DetectedByDispatchMarker() { - // The same SUB_WORKFLOW type is a genuine tool call when the dispatch - // script marked it as one. + // Same type, but the dispatch script marked it, so it is a real tool call. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"SUB_WORKFLOW","taskDefName":"billing_agent_wf", @@ -354,8 +335,7 @@ public async Task Events_FailedRun_EndsWithErrorNotDone() [Fact] public async Task HttpToolResult_FallsBackToWholeOutput() { - // An HTTP tool answers under `response`, not `result`, so keying only on - // `result` would report a tool call with no result at all. + // An HTTP tool answers under `response`, so keying only on `result` loses it. var result = await WaitWithTasksAsync(""" {"tasks":[ {"taskType":"HTTP","taskDefName":"lookup_order","referenceTaskName":"lookup_order_0", diff --git a/Conductor.AI/Result.cs b/Conductor.AI/Result.cs index 3a5c12bd..4ed14ddc 100644 --- a/Conductor.AI/Result.cs +++ b/Conductor.AI/Result.cs @@ -305,9 +305,7 @@ public async Task WaitAsync(CancellationToken cancellationToken = d { // Fetch full execution record for token usage and finish reason var execution = await _http.GetExecutionAsync(_executionId, cancellationToken); - // Walk the workflow's tasks to recover the tool calls and the - // events the run produced (enrichment read — null is fine, - // and yields a result carrying neither). + // Tool calls and events come from the task list (enrichment read; null is fine). var workflowWithTasks = await _http.GetWorkflowWithTasksAsync(_executionId, cancellationToken); return BuildResult(status!, s, execution, workflowWithTasks); } @@ -561,27 +559,10 @@ private static AgentResult BuildResult( } // ── Tool-call extraction ───────────────────────────────────────── - // - // A tool task is identified by its Conductor task type, allowlisted off the - // server's ToolCompiler.TYPE_MAP — for the types the agent layer also uses - // for its own structure, and for the worker kind, corroborated by a dispatch - // marker. It is never identified by its reference task name: the server - // seeds that from the provider's tool-call id (OpenAI's `call_...`, - // Anthropic's `toolu_...`, else a UUID) and appends the fork index and loop - // iteration, so a prefix test only ever matches one provider. - // - // The tool's real name is likewise never the task type. Conductor sets an - // executed SIMPLE task's type to the task's own name, which is the tool - // name for a worker tool and is why that one kind used to read correctly; - // every other kind carries its system task type there instead. /// - /// Task types only a tool compiles to, from the server's - /// ToolCompiler.TYPE_MAP — plus GENERATE_PDF, which that map - /// omits although the server compiles a generate_pdf tool to it. - /// SIMPLE is absent deliberately: a worker tool's executed task type - /// is the tool's own name, so that kind is recognised by the dispatch - /// markers below rather than by type. + /// Task types only a tool compiles to: the server's ToolCompiler.TYPE_MAP. + /// GENERATE_PDF is absent from that map, but media tools compile to it. /// private static readonly HashSet ToolTaskTypes = new(StringComparer.Ordinal) { @@ -590,52 +571,22 @@ private static AgentResult BuildResult( "LLM_INDEX_TEXT", "LLM_SEARCH_INDEX", "PULL_WORKFLOW_MESSAGES", }; - /// - /// Task types a tool compiles to that the agent layer also uses for its own - /// structure — SUB_WORKFLOW for a sub-agent, a strategy workflow, a - /// router and a plan execution; HUMAN for a plan's approval step. - /// The type alone therefore proves nothing, and a dispatch marker is - /// required as well. - /// + /// Tool task types the agent layer also emits for sub-agents, routers and plan approval. private static readonly HashSet AmbiguousToolTaskTypes = new(StringComparer.Ordinal) { "SUB_WORKFLOW", "HUMAN", }; - /// - /// Input keys the server's tool-dispatch script injects. - /// _agent_tool_name is set for every tool kind on the static - /// dispatch path; _agent_state is set for worker tools on both the - /// static and the dynamic-tools path. - /// + /// Markers the server's dispatch script injects on a tool task it dispatched. private const string AgentToolNameKey = "_agent_tool_name"; private const string AgentStateKey = "_agent_state"; - /// The dispatch method name — the tool name on the dynamic-tools path. + /// The tool name on the dynamic-tools path, which injects no _agent_tool_name. private const string MethodKey = "method"; /// - /// Whether a task is an LLM-dispatched tool call: an unambiguous tool task - /// type, or a type that needs corroborating evidence that the LLM dispatched - /// it. - /// - /// Two cases need that evidence. The worker case, because a worker tool's - /// task type is the tool's own name and so cannot be allowlisted; what marks - /// it is Conductor setting an executed SIMPLE task's type to the task's own - /// name, which is also its taskDefName — a signature no system task - /// shares. And the ambiguous types above, which the agent layer emits for - /// its own structure far more often than for a tool. - /// - /// Neither half of the test suffices alone: a multi-agent - /// SET_VARIABLE task carries _agent_state without being a tool - /// call, and a multi-agent handoff is a SUB_WORKFLOW without being - /// one either. - /// - /// Accepted cost: the dynamic-tools dispatch path sets no marker on a - /// non-worker tool, so an agent-as-tool or human tool dispatched that way is - /// missed. That is the right way to be wrong — the alternative reports a - /// fabricated tool call for every handoff in every multi-agent run, and a - /// tool call that did not happen is worse than one that is absent. + /// Whether a task is a tool call the LLM dispatched. Never judged by reference + /// name, which carries the provider's tool-call id. /// private static bool IsToolTask(JsonNode task) { @@ -643,6 +594,11 @@ private static bool IsToolTask(JsonNode task) if (taskType is null) return false; if (ToolTaskTypes.Contains(taskType)) return true; + // A worker tool's task type is the tool's own name, so only the marker can + // identify it. The marker also keeps the agent's own structure out: on type + // alone every handoff would read as a tool call that never happened. Cost is + // that the dynamic-tools path marks no non-worker tool, so an agent-as-tool + // dispatched there is missed. var needsMarker = AmbiguousToolTaskTypes.Contains(taskType) || taskType == task["taskDefName"]?.GetValue(); return needsMarker @@ -651,9 +607,8 @@ private static bool IsToolTask(JsonNode task) } /// - /// Resolve a tool task's tool name: inputData._agent_tool_name, then - /// inputData.method (the dynamic-tools dispatch path sets no - /// _agent_tool_name), then taskDefName. + /// The tool's name from the payload, never the task type, which holds the tool + /// name only for a worker tool. /// private static string ResolveToolName(JsonNode task) { @@ -671,7 +626,7 @@ private static string ResolveToolName(JsonNode task) ? s : null; - /// The tool's arguments — the task's input with the runtime's own keys stripped. + /// The tool's arguments: the task's input with the runtime's own keys stripped. private static Dictionary ToolArgs(JsonObject inputData) { var cleaned = new Dictionary(); @@ -687,10 +642,8 @@ private static Dictionary ToolArgs(JsonObject inputData) } /// - /// The tool's result: the task's result output, or its whole output - /// when there is no such key — an HTTP tool answers under response, - /// so keying only on result would report no result at all for it. - /// Matches the fallback the server's AgentEventListener applies. + /// The task's result, or its whole output when there is none: an HTTP tool + /// answers under response. Matches the server's AgentEventListener. /// private static object? ToolResult(JsonObject outputData) { @@ -701,22 +654,8 @@ private static Dictionary ToolArgs(JsonObject inputData) } /// - /// Walk the workflow's tasks once and recover, per LLM-dispatched tool call, - /// both the entry (name, arguments, - /// result) and the / - /// event pair — two views of the same - /// call, so they are built together and cannot drift apart. - /// - /// The returned event list carries the tool events only; the caller appends - /// the terminal event. It is never null, so enumerating - /// never throws. Tool calls stay null when - /// there were none, as they always have. - /// - /// This is a reconstruction from completed tasks, not the stream - /// delivers live: the server emits events for - /// thinking steps, failed tasks, handoffs and guardrails, none of which - /// survives into the terminal record. Use when the - /// events themselves are the point. + /// Both views of each tool call in one pass. Reconstructed from finished tasks, + /// so narrower than the live stream. /// private static (List>? ToolCalls, List Events) ExtractToolActivity( JsonNode? workflowWithTasks, string executionId) From c9b0da19d101a82eb10be1f66b1892062cdcf5fb Mon Sep 17 00:00:00 2001 From: francisco-orkes Date: Tue, 8 Sep 2026 21:57:08 -0700 Subject: [PATCH 3/3] fix(agents): report a dispatched tool call that recorded no output Review follow-ups on the tool-extraction change. ExtractToolActivity skipped any task without outputData before deciding whether it was a tool task at all, so a tool the LLM dispatched that failed or never finished left no ToolCalls entry and no ToolCall event. The call demonstrably happened; it is now reported with its name and arguments and no result. Servers send outputData as an empty object rather than omitting it, so this is a narrow path, but the ordering hid it rather than handling it. AmbiguousToolTaskTypes named the ambiguity rather than the rule it encodes and becomes MarkerRequiredToolTaskTypes, which is what the call site asks. The internal-input-key list joins the two dispatch markers as a named set. Test fixtures built by positional Replace over JSON, which would rewrite any later occurrence of TYPE or NAME, become interpolated raw strings. Docs: the new concepts section leads with its example, per docs/documentation-standard.md rule 1, and the reference entry links rather than repeating the gloss, per rule 5. --- .../AgentToolCallExtractionTests.cs | 46 +++++++++++++++---- Conductor.AI/Result.cs | 23 ++++++---- docs/agents/concepts/streaming-hitl.md | 5 ++ docs/agents/reference/api.md | 5 +- 4 files changed, 58 insertions(+), 21 deletions(-) diff --git a/Conductor.AI.Tests/AgentToolCallExtractionTests.cs b/Conductor.AI.Tests/AgentToolCallExtractionTests.cs index c27f94fb..81a0104b 100644 --- a/Conductor.AI.Tests/AgentToolCallExtractionTests.cs +++ b/Conductor.AI.Tests/AgentToolCallExtractionTests.cs @@ -84,13 +84,15 @@ public async Task WorkerTool_NameFromAgentToolName() [InlineData("PULL_WORKFLOW_MESSAGES", "read_queue")] public async Task SystemTaskTool_NameFromAgentToolName_NotTaskType(string taskType, string toolName) { - var result = await WaitWithTasksAsync(""" + var result = await WaitWithTasksAsync($$""" {"tasks":[ - {"taskType":"TYPE","taskDefName":"NAME","referenceTaskName":"NAME_0__1", - "inputData":{"_agent_tool_name":"NAME","query":"x"}, - "outputData":{"result":"ok"}} + {"taskType":"{{taskType}}","taskDefName":"{{toolName}}", + "referenceTaskName":"{{toolName}}_0__1", + "inputData":{"_agent_tool_name":"{{toolName}}","query":"x"}, + "outputData":{"result":"ok"} + } ]} - """.Replace("TYPE", taskType).Replace("NAME", toolName)); + """); var toolCall = Assert.Single(result.ToolCalls!); Assert.Equal(toolName, toolCall["name"]); @@ -230,14 +232,17 @@ public async Task WorkerTaskWithoutDispatchMarker_NotAToolCall() [InlineData("HUMAN", "plan_approval")] public async Task AgentStructureReusingAToolTaskType_NotAToolCall(string taskType, string taskDefName) { - var result = await WaitWithTasksAsync(""" + var result = await WaitWithTasksAsync($$""" {"tasks":[ - {"taskType":"TYPE","taskDefName":"NAME","referenceTaskName":"0_NAME__1", + {"taskType":"{{taskType}}","taskDefName":"{{taskDefName}}", + "referenceTaskName":"0_{{taskDefName}}__1", "inputData":{"prompt":"hello","media":[],"session_id":"s1", - "context":{"turn":1}}, - "outputData":{"result":"handled"}} + "context":{"turn":1} + }, + "outputData":{"result":"handled"} + } ]} - """.Replace("TYPE", taskType).Replace("NAME", taskDefName)); + """); Assert.Null(result.ToolCalls); Assert.Single(result.Events!); @@ -332,6 +337,27 @@ public async Task Events_FailedRun_EndsWithErrorNotDone() Assert.Equal("tool worker never polled", terminal.Content); } + [Fact] + public async Task ToolTaskWithNoOutput_StillReported_WithoutAResult() + { + // A tool the LLM dispatched that failed or never finished records no output. + // Dropping it would hide a call that demonstrably happened. + var result = await WaitWithTasksAsync(""" + {"tasks":[ + {"taskType":"HTTP","taskDefName":"lookup_order","referenceTaskName":"lookup_order_0", + "inputData":{"_agent_tool_name":"lookup_order","order":"A-1"}, + "outputData":{}} + ]} + """); + + var toolCall = Assert.Single(result.ToolCalls!); + Assert.Equal("lookup_order", toolCall["name"]); + Assert.False(toolCall.ContainsKey("result")); + Assert.Equal( + [EventType.ToolCall, EventType.ToolResult, EventType.Done], + result.Events!.Select(e => e.Type)); + } + [Fact] public async Task HttpToolResult_FallsBackToWholeOutput() { diff --git a/Conductor.AI/Result.cs b/Conductor.AI/Result.cs index 4ed14ddc..0cd2c09e 100644 --- a/Conductor.AI/Result.cs +++ b/Conductor.AI/Result.cs @@ -572,7 +572,7 @@ private static AgentResult BuildResult( }; /// Tool task types the agent layer also emits for sub-agents, routers and plan approval. - private static readonly HashSet AmbiguousToolTaskTypes = new(StringComparer.Ordinal) + private static readonly HashSet MarkerRequiredToolTaskTypes = new(StringComparer.Ordinal) { "SUB_WORKFLOW", "HUMAN", }; @@ -584,6 +584,12 @@ private static AgentResult BuildResult( /// The tool name on the dynamic-tools path, which injects no _agent_tool_name. private const string MethodKey = "method"; + /// Runtime keys carried on a tool task's input that are not the tool's arguments. + private static readonly HashSet InternalInputKeys = new(StringComparer.Ordinal) + { + MethodKey, "evaluatorType", "expression", "ctx", "workerTag", "agentConfig", + }; + /// /// Whether a task is a tool call the LLM dispatched. Never judged by reference /// name, which carries the provider's tool-call id. @@ -599,7 +605,7 @@ private static bool IsToolTask(JsonNode task) // alone every handoff would read as a tool call that never happened. Cost is // that the dynamic-tools path marks no non-worker tool, so an agent-as-tool // dispatched there is missed. - var needsMarker = AmbiguousToolTaskTypes.Contains(taskType) + var needsMarker = MarkerRequiredToolTaskTypes.Contains(taskType) || taskType == task["taskDefName"]?.GetValue(); return needsMarker && task["inputData"] is JsonObject inputData @@ -633,9 +639,7 @@ private static Dictionary ToolArgs(JsonObject inputData) foreach (var kv in inputData) { var k = kv.Key; - if (k.StartsWith('_') || k == MethodKey || k is "evaluatorType" or "expression" or "ctx" - or "workerTag" or "agentConfig") - continue; + if (k.StartsWith('_') || InternalInputKeys.Contains(k)) continue; cleaned[k] = JsonSerializer.Deserialize(kv.Value?.ToJsonString() ?? "null", ConductorAgentJson.Options)!; } return cleaned; @@ -667,12 +671,15 @@ private static (List>? ToolCalls, List Ev foreach (var task in tasks) { - if (task is null || task["outputData"] is not JsonObject outputData || !IsToolTask(task)) - continue; + if (task is null || !IsToolTask(task)) continue; var name = ResolveToolName(task); var args = task["inputData"] is JsonObject inputData ? ToolArgs(inputData) : null; - var result = ToolResult(outputData); + // A task that recorded no output still ran: report the call without a result + // rather than dropping it, which would hide a tool call that failed. + var result = task["outputData"] is JsonObject { Count: > 0 } outputData + ? ToolResult(outputData) + : null; var tc = new Dictionary { ["name"] = name }; if (args is not null) tc["args"] = args; diff --git a/docs/agents/concepts/streaming-hitl.md b/docs/agents/concepts/streaming-hitl.md index c2c231eb..cf653b89 100644 --- a/docs/agents/concepts/streaming-hitl.md +++ b/docs/agents/concepts/streaming-hitl.md @@ -31,6 +31,11 @@ Event types: `Thinking`, `ToolCall`, `ToolResult`, `GuardrailPass`, ### Events on a waited result +```csharp +var result = await handle.WaitAsync(); +foreach (var ev in result.Events!) Console.WriteLine($"{ev.Type} {ev.ToolName}"); +``` + `AgentResult.Events` carries the run's tool activity — a `ToolCall`/`ToolResult` pair per tool call, closed by a terminal `Done`, or `Error` for a run that did not complete. It is reconstructed from the finished execution's tasks, so it is never diff --git a/docs/agents/reference/api.md b/docs/agents/reference/api.md index bb749b2c..ecac78cf 100644 --- a/docs/agents/reference/api.md +++ b/docs/agents/reference/api.md @@ -211,9 +211,8 @@ Positions map to server task names: `before_agent`, `after_agent`, `before_model **`AgentResult`** (record): `ExecutionId`, `CorrelationId`, `Output` (`Dictionary?`; final text usually `Output["result"]`), `Messages`, -`ToolCalls`, `Status`, `FinishReason`, `Error`, `TokenUsage`, `Metadata`, `Events` -(tool activity plus a terminal event, reconstructed from the execution's tasks — -see [streaming-hitl.md](../concepts/streaming-hitl.md#events-on-a-waited-result)), +`ToolCalls`, `Status`, `FinishReason`, `Error`, `TokenUsage`, `Metadata`, +`Events` (see [streaming-hitl.md](../concepts/streaming-hitl.md#events-on-a-waited-result)), `SubResults`. Convenience: `IsSuccess`, `IsFailed`, `IsRejected`, `PrintResult()`. **`AgentHandle`** — `ExecutionId`, `RunId`, `WaitAsync(ct)`, `StreamAsync(ct)`,