From 53073332fa714fe369774bcf35ee373e08480fcc Mon Sep 17 00:00:00 2001 From: David Gageot Date: Thu, 1 Oct 2026 10:16:07 +0200 Subject: [PATCH 1/2] feat: add debug tool command to call tools directly Lets you invoke an agent's tool from the CLI with a JSON parameters object, bypassing the LLM turn. Supports deferred tools without activation and agent selection via --agent. Assisted-By: cagent --- cmd/root/debug.go | 20 ++- cmd/root/debug_test.go | 10 +- cmd/root/debug_tool.go | 102 +++++++++++ cmd/root/debug_tool_test.go | 344 ++++++++++++++++++++++++++++++++++++ docs/features/cli/index.md | 12 +- 5 files changed, 482 insertions(+), 6 deletions(-) create mode 100644 cmd/root/debug_tool.go create mode 100644 cmd/root/debug_tool_test.go diff --git a/cmd/root/debug.go b/cmd/root/debug.go index 324516f967..9d35c59109 100644 --- a/cmd/root/debug.go +++ b/cmd/root/debug.go @@ -25,6 +25,8 @@ import ( type debugFlags struct { modelOverrides []string toolsetsJSON bool + toolJSON bool + toolAgent string skillsJSON bool runConfig config.RuntimeConfig } @@ -82,6 +84,22 @@ func newDebugCmd() *cobra.Command { } toolsetsCmd.Flags().BoolVar(&flags.toolsetsJSON, "json", false, "Output in JSON format") cmd.AddCommand(toolsetsCmd) + toolCmd := &cobra.Command{ + Use: "tool | [parameters-json]", + Short: "Call a tool of an agent directly", + Long: "Call a tool of an agent directly, without an LLM turn.\n\n" + + "Parameters must be a JSON object (defaults to {}). Use --agent to select an agent.\n" + + "Use 'debug toolsets --json' to inspect tool names and parameter schemas.\n\n" + + "Calls have real side effects and bypass session hooks and approval checks.\n" + + "Tools that require an agent runtime are not supported.", + Example: ` docker agent debug tool agent.yaml read_file '{"path":"README.md"}' + docker agent debug tool agent.yaml shell '{"cmd":"pwd"}' --agent root --json`, + Args: cobra.RangeArgs(2, 3), + RunE: flags.runDebugToolCommand, + } + toolCmd.Flags().StringVarP(&flags.toolAgent, "agent", "a", "", "Name of the agent (defaults to the team's default agent)") + toolCmd.Flags().BoolVar(&flags.toolJSON, "json", false, "Output the full tool result in JSON format") + cmd.AddCommand(toolCmd) skillsCmd := &cobra.Command{ Use: "skills |", Short: "Debug the skills of an agent", @@ -174,7 +192,7 @@ func (f *debugFlags) runDebugToolsetsCommand(cmd *cobra.Command, args []string) continue } - agentTools, err := agent.Tools(ctx) + agentTools, err := agent.ToolsWithCatalog(ctx) if err != nil { slog.ErrorContext(ctx, "Failed to query tools", "name", agent.Name(), "error", err) continue diff --git a/cmd/root/debug_test.go b/cmd/root/debug_test.go index ab6103b7fe..9d132f14df 100644 --- a/cmd/root/debug_test.go +++ b/cmd/root/debug_test.go @@ -43,9 +43,8 @@ func TestDebug_VisibleInAdvancedGroup(t *testing.T) { assert.Equal(t, "advanced", cmd.GroupID) } -// Non-regression: `toolsets --json` and `skills --json` must not share flag -// storage, otherwise running one with --json on a reused command tree makes -// the other emit JSON too. +// Non-regression: debug subcommands must not share JSON flag storage, otherwise +// reusing a command tree makes unrelated subcommands emit JSON too. func TestDebug_JSONFlagsAreIndependent(t *testing.T) { t.Parallel() @@ -54,6 +53,8 @@ func TestDebug_JSONFlagsAreIndependent(t *testing.T) { require.NoError(t, err) skillsCmd, _, err := cmd.Find([]string{"skills"}) require.NoError(t, err) + toolCmd, _, err := cmd.Find([]string{"tool"}) + require.NoError(t, err) require.NoError(t, toolsetsCmd.Flags().Set("json", "true")) @@ -64,6 +65,9 @@ func TestDebug_JSONFlagsAreIndependent(t *testing.T) { assert.True(t, toolsetsJSON) assert.False(t, skillsJSON) + toolJSON, err := toolCmd.Flags().GetBool("json") + require.NoError(t, err) + assert.False(t, toolJSON) } const flavoredConfig = ` diff --git a/cmd/root/debug_tool.go b/cmd/root/debug_tool.go new file mode 100644 index 0000000000..d5fa973317 --- /dev/null +++ b/cmd/root/debug_tool.go @@ -0,0 +1,102 @@ +package root + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "slices" + + "github.com/spf13/cobra" + + "github.com/docker/docker-agent/pkg/telemetry" + "github.com/docker/docker-agent/pkg/tools" +) + +func (f *debugFlags) runDebugToolCommand(cmd *cobra.Command, args []string) (commandErr error) { + ctx := cmd.Context() + // Tool parameters may contain secrets; keep them out of command telemetry. + telemetry.TrackCommand(ctx, "debug", []string{"tool"}) + defer func() { + if commandErr != nil { + telemetry.TrackCommandError(ctx, "debug", []string{"tool"}, errors.New("tool invocation failed")) + } + }() + + arguments := "{}" + if len(args) == 3 { + arguments = args[2] + } + var params map[string]json.RawMessage + if err := json.Unmarshal([]byte(arguments), ¶ms); err != nil { + return fmt.Errorf("parameters must be a JSON object: %w", err) + } + if params == nil { + return errors.New("parameters must be a JSON object, not null") + } + + t, err := f.loadTeam(ctx, args[0]) + if err != nil { + return err + } + defer stopToolSets(ctx, t) + + agent, err := t.AgentOrDefault(f.toolAgent) + if err != nil { + return err + } + + // Include deferred tools: activation would otherwise be lost between CLI calls. + available, err := agent.ToolsWithCatalog(ctx) + for _, warning := range agent.DrainWarnings() { + fmt.Fprintln(cmd.ErrOrStderr(), "Warning:", warning) + } + if err != nil { + return fmt.Errorf("listing tools for agent %q: %w", agent.Name(), err) + } + + index := slices.IndexFunc(available, func(tool tools.Tool) bool { return tool.Name == args[1] }) + if index < 0 { + return fmt.Errorf("tool %q not found for agent %q; use 'debug toolsets --json' to list tools", args[1], agent.Name()) + } + + result, err := callDebugTool(ctx, available[index], arguments) + if err != nil { + return err + } + + if f.toolJSON { + err = encodeJSON(cmd, result) + } else { + _, err = fmt.Fprintln(cmd.OutOrStdout(), result.Output) + } + if err != nil { + return err + } + if result.IsError { + return fmt.Errorf("tool %q returned an error for agent %q", args[1], agent.Name()) + } + return nil +} + +func callDebugTool(ctx context.Context, tool tools.Tool, arguments string) (*tools.ToolCallResult, error) { + if tool.RuntimeHandler != "" || tool.Handler == nil { + return nil, fmt.Errorf("tool %q requires an agent runtime and cannot be called directly", tool.Name) + } + toolCall := tools.ToolCall{ + ID: "debug_" + tool.Name, + Type: "function", + Function: tools.FunctionCall{ + Name: tool.Name, + Arguments: arguments, + }, + } + result, err := tool.Handler(ctx, toolCall, tools.NopRuntime{}) + if err != nil { + return nil, fmt.Errorf("calling tool %q: %w", tool.Name, err) + } + if result == nil { + return nil, fmt.Errorf("tool %q returned no result", tool.Name) + } + return result, nil +} diff --git a/cmd/root/debug_tool_test.go b/cmd/root/debug_tool_test.go new file mode 100644 index 0000000000..0a30337a4e --- /dev/null +++ b/cmd/root/debug_tool_test.go @@ -0,0 +1,344 @@ +package root + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "testing" + + "github.com/modelcontextprotocol/go-sdk/mcp" + "github.com/spf13/cobra" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/docker/docker-agent/pkg/environment" + "github.com/docker/docker-agent/pkg/tools" +) + +const debugToolConfig = ` +agents: + root: + model: test + toolsets: + - type: think + - type: filesystem + tools: [read_file] + helper: + model: test + toolsets: + - type: shell + defer: true +models: + test: + provider: openai + model: gpt-4o + max_tokens: 100 +` + +func runDebugTool(t *testing.T, flags *debugFlags, args ...string) (string, error) { + t.Helper() + + return runDebugToolConfig(t, flags, debugToolConfig, args...) +} + +func runDebugToolConfig(t *testing.T, flags *debugFlags, cfg string, args ...string) (string, error) { + t.Helper() + + dir := t.TempDir() + path := filepath.Join(dir, "agent.yaml") + require.NoError(t, os.WriteFile(path, []byte(cfg), 0o600)) + require.NoError(t, os.WriteFile(filepath.Join(dir, "hello.txt"), []byte("hello from a tool"), 0o600)) + flags.runConfig.WorkingDir = dir + flags.runConfig.EnvProviderOverride = environment.NewMapEnvProvider(map[string]string{"OPENAI_API_KEY": "test-key"}) + + var out bytes.Buffer + cmd := &cobra.Command{} + cmd.SetOut(&out) + cmd.SetErr(&bytes.Buffer{}) + cmd.SetContext(t.Context()) + + err := flags.runDebugToolCommand(cmd, append([]string{path}, args...)) + return out.String(), err +} + +func TestDebugToolCommand(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + flags debugFlags + args []string + output string + err string + }{ + { + name: "default agent", + args: []string{"think", `{"thought":"testing tools"}`}, + output: "Thoughts:\ntesting tools\n", + }, + { + name: "default parameters", + args: []string{"think"}, + output: "Thoughts:\n\n", + }, + { + name: "working directory", + args: []string{"read_file", `{"path":"hello.txt"}`}, + output: "hello from a tool", + }, + { + name: "selected agent and deferred tool", + flags: debugFlags{toolAgent: "helper"}, + args: []string{"shell", `{"cmd":"echo hello"}`}, + output: "hello\n", + }, + { + name: "unknown agent", + flags: debugFlags{toolAgent: "missing"}, + args: []string{"think"}, + err: "agent not found: missing (available agents: root, helper)", + }, + { + name: "tool belongs to another agent", + args: []string{"shell"}, + err: `tool "shell" not found for agent "root"`, + }, + { + name: "filtered tool", + args: []string{"write_file", `{"path":"hello.txt","content":"overwrite"}`}, + err: `tool "write_file" not found for agent "root"`, + }, + { + name: "invalid JSON", + args: []string{"think", `{"thought":`}, + err: "parameters must be a JSON object", + }, + { + name: "array parameters", + args: []string{"think", `[]`}, + err: "parameters must be a JSON object", + }, + { + name: "null parameters", + args: []string{"think", `null`}, + err: "parameters must be a JSON object, not null", + }, + { + name: "invalid parameter type", + args: []string{"think", `{"thought":{}}`}, + err: `calling tool "think"`, + }, + { + name: "tool error", + args: []string{"read_file", `{"path":"missing.txt"}`}, + output: "not found\n", + err: `tool "read_file" returned an error for agent "root"`, + }, + } + for i := range tests { + tt := &tests[i] + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + + out, err := runDebugTool(t, &tt.flags, tt.args...) + if tt.err != "" { + require.ErrorContains(t, err, tt.err) + } else { + require.NoError(t, err) + } + if tt.output != "" { + assert.Contains(t, out, tt.output) + } else { + assert.Empty(t, out) + } + }) + } +} + +func TestDebugToolCommand_JSON(t *testing.T) { + t.Parallel() + + for _, isError := range []bool{false, true} { + t.Run(map[bool]string{false: "success", true: "tool error"}[isError], func(t *testing.T) { + t.Parallel() + + args := []string{"think", `{"thought":"hello"}`} + if isError { + args = []string{"read_file", `{"path":"missing.txt"}`} + } + out, err := runDebugTool(t, &debugFlags{toolJSON: true}, args...) + if isError { + require.Error(t, err) + } else { + require.NoError(t, err) + } + var result tools.ToolCallResult + require.NoError(t, json.Unmarshal([]byte(out), &result)) + assert.Equal(t, isError, result.IsError) + assert.NotEmpty(t, result.Output) + }) + } +} + +func TestDebugToolCommand_Arguments(t *testing.T) { + t.Parallel() + + cmd, _, err := newDebugCmd().Find([]string{"tool"}) + var out bytes.Buffer + cmd.SetOut(&out) + require.NoError(t, err) + require.Equal(t, "tool", cmd.Name()) + for _, args := range [][]string{nil, {"agent.yaml"}, {"agent.yaml", "think", "{}", "extra"}} { + require.Error(t, cmd.Args(cmd, args)) + } + assert.NoError(t, cmd.Args(cmd, []string{"agent.yaml", "think"})) + assert.NoError(t, cmd.Args(cmd, []string{"agent.yaml", "think", "{}"})) + require.NoError(t, cmd.ParseFlags([]string{"-a", "helper", "--json"})) + assert.NoError(t, cmd.Help()) + assert.Contains(t, out.String(), "Call a tool of an agent directly") +} + +func TestCallDebugTool(t *testing.T) { + t.Parallel() + + t.Run("call shape and full result", func(t *testing.T) { + t.Parallel() + + expected := &tools.ToolCallResult{ + Output: "result", + Images: []tools.MediaContent{{Data: "aGVsbG8=", MimeType: "image/png"}}, + StructuredContent: map[string]any{"key": "value"}, + } + tool := tools.Tool{Name: "test", Handler: func(ctx context.Context, tc tools.ToolCall, rt tools.Runtime) (*tools.ToolCallResult, error) { + assert.Equal(t, t.Context(), ctx) + assert.Equal(t, "debug_test", tc.ID) + assert.Equal(t, tools.ToolType("function"), tc.Type) + assert.Equal(t, tools.FunctionCall{Name: "test", Arguments: `{"value":42}`}, tc.Function) + assert.IsType(t, tools.NopRuntime{}, rt) + return expected, nil + }} + result, err := callDebugTool(t.Context(), tool, `{"value":42}`) + require.NoError(t, err) + assert.Same(t, expected, result) + }) + + t.Run("handler error", func(t *testing.T) { + t.Parallel() + + expected := errors.New("handler failed") + tool := tools.Tool{Name: "test", Handler: func(context.Context, tools.ToolCall, tools.Runtime) (*tools.ToolCallResult, error) { + return nil, expected + }} + _, err := callDebugTool(t.Context(), tool, "{}") + require.ErrorIs(t, err, expected) + }) + + t.Run("nil result", func(t *testing.T) { + t.Parallel() + + tool := tools.Tool{Name: "test", Handler: func(context.Context, tools.ToolCall, tools.Runtime) (*tools.ToolCallResult, error) { + return nil, nil + }} + _, err := callDebugTool(t.Context(), tool, "{}") + require.ErrorContains(t, err, "returned no result") + }) + + t.Run("runtime tool", func(t *testing.T) { + t.Parallel() + + tool := tools.Tool{Name: "test", RuntimeHandler: "runtime", Handler: func(context.Context, tools.ToolCall, tools.Runtime) (*tools.ToolCallResult, error) { + t.Fatal("runtime handler must not be called") + return nil, nil + }} + _, err := callDebugTool(t.Context(), tool, "{}") + require.ErrorContains(t, err, "requires an agent runtime") + }) + + t.Run("missing handler", func(t *testing.T) { + t.Parallel() + + _, err := callDebugTool(t.Context(), tools.Tool{Name: "test"}, "{}") + require.ErrorContains(t, err, "requires an agent runtime") + }) +} + +func TestDebugToolCommand_MCP(t *testing.T) { + t.Parallel() + + server := mcp.NewServer(&mcp.Implementation{Name: "debug-test", Version: "1.0.0"}, nil) + mcp.AddTool(server, &mcp.Tool{Name: "echo"}, func(_ context.Context, _ *mcp.CallToolRequest, args struct { + Message string `json:"message"` + }, + ) (*mcp.CallToolResult, any, error) { + return &mcp.CallToolResult{ + Content: []mcp.Content{ + &mcp.TextContent{Text: args.Message}, + &mcp.ImageContent{Data: []byte("image"), MIMEType: "image/png"}, + }, + }, map[string]string{"message": args.Message}, nil + }) + httpServer := httptest.NewServer(mcp.NewStreamableHTTPHandler(func(*http.Request) *mcp.Server { + return server + }, nil)) + t.Cleanup(httpServer.Close) + + cfg := fmt.Sprintf(`agents: + root: + model: test + toolsets: + - type: mcp + name: test + allow_private_ips: true + remote: + url: %s + transport_type: streamable +models: + test: + provider: openai + model: gpt-4o + max_tokens: 100 +`, httpServer.URL) + out, err := runDebugToolConfig(t, &debugFlags{toolJSON: true}, cfg, "test_echo", `{"message":"hello MCP"}`) + require.NoError(t, err) + var result tools.ToolCallResult + require.NoError(t, json.Unmarshal([]byte(out), &result)) + assert.Equal(t, "hello MCP", result.Output) + assert.False(t, result.IsError) + assert.Equal(t, []tools.MediaContent{{Data: "aW1hZ2U=", MimeType: "image/png"}}, result.Images) + assert.Equal(t, map[string]any{"message": "hello MCP"}, result.StructuredContent) +} + +func TestDebugToolsetsCommand_IncludesDeferredTools(t *testing.T) { + t.Parallel() + + path := filepath.Join(t.TempDir(), "agent.yaml") + require.NoError(t, os.WriteFile(path, []byte(debugToolConfig), 0o600)) + flags := &debugFlags{toolsetsJSON: true} + flags.runConfig.EnvProviderOverride = environment.NewMapEnvProvider(map[string]string{"OPENAI_API_KEY": "test-key"}) + var out bytes.Buffer + cmd := &cobra.Command{} + cmd.SetContext(t.Context()) + cmd.SetOut(&out) + require.NoError(t, flags.runDebugToolsetsCommand(cmd, []string{path})) + var infos []agentToolsInfo + require.NoError(t, json.Unmarshal(out.Bytes(), &infos)) + require.Len(t, infos, 2) + for _, info := range infos { + if info.Agent != "helper" { + continue + } + for _, tool := range info.Tools { + if tool.Name == "shell" { + assert.NotNil(t, tool.Parameters) + return + } + } + } + t.Fatal("deferred shell tool must be discoverable") +} diff --git a/docs/features/cli/index.md b/docs/features/cli/index.md index 8392db975a..57626f167c 100644 --- a/docs/features/cli/index.md +++ b/docs/features/cli/index.md @@ -749,7 +749,8 @@ $ docker agent debug [flags] | Subcommand | Description | | ---------- | ----------- | | `config [flavor...]` | Print the fully-resolved, canonical form of an agent's configuration (defaults applied, references resolved). When [flavors](../../configuration/flavors/index.md) are given they are applied in order and the `flavors` section is dropped from the output. | -| `toolsets ` | List every toolset each agent in the config exposes, with each tool's name and description. Add `--json` for machine-readable output including each tool's parameters, annotations, and output schema. | +| `toolsets ` | List every toolset each agent in the config exposes, with each tool's name and description. Add `--json` for machine-readable output including each tool's parameters, annotations, and output schema. Deferred tools are included. | +| `tool [parameters-json]` | Call a tool directly with a JSON object (defaults to `{}`), without an LLM turn. Select an agent with `-a` / `--agent`; otherwise the team's default agent is used. Add `--json` for the full result, including structured content and media. Deferred tools can be called without activation. | | `skills ` | List the skills discovered for each agent, marking forked skills. Add `--json` for machine-readable output; each skill includes a `path` field when it is backed by a file (omitted for inline skills). | | `title ` | Generate a session title for `` using the same title-generation path the TUI uses (including any configured `title_model`), without starting a session. See [Session Titles](../sessions/index.md#session-titles). | | `auth` | Print parsed Docker authentication info from the token in use (source, subject, issuer, expiry, username/email). Add `--json` for machine-readable output. | @@ -763,6 +764,8 @@ $ docker agent debug config agent.yaml $ docker agent debug config agent.yaml cheap with-shell $ docker agent debug toolsets agent.yaml $ docker agent debug toolsets agent.yaml --json +$ docker agent debug tool agent.yaml read_file '{"path":"README.md"}' +$ docker agent debug tool agent.yaml shell '{"cmd":"pwd"}' -a root --json $ docker agent debug skills agent.yaml $ docker agent debug skills agent.yaml --json $ docker agent debug title agent.yaml "How do I configure a fallback model?" @@ -771,6 +774,11 @@ $ docker agent debug oauth list $ docker agent debug oauth login agent.yaml github ``` +> [!WARNING] +> **`debug tool` executes real tool calls** +> +> Calls can modify files, run commands, or contact external services. They bypass session hooks and approval checks. Tools that require an agent runtime (such as delegation and handoff) are not supported. Tool errors print their result and exit with a non-zero status. Use `debug toolsets --json` to inspect parameter schemas before calling a tool. + > [!WARNING] > **`debug auth --json` prints the full bearer token** > @@ -778,7 +786,7 @@ $ docker agent debug oauth login agent.yaml github The `Source` field says where the token came from: `docker desktop`, or `minted from the stored access token` when it was obtained by exchanging the access token `docker login` stored. See [Docker authentication](../../guides/secrets/index.md#docker-authentication). -The `config`, `toolsets`, `skills`, and `title` subcommands also accept [runtime configuration flags](#runtime-configuration-flags) (`--working-dir`, `--models-gateway`, …); `title` additionally accepts `--model` to override the model used to resolve the config before generating the title. +The `config`, `toolsets`, `tool`, `skills`, and `title` subcommands also accept [runtime configuration flags](#runtime-configuration-flags) (`--working-dir`, `--models-gateway`, …); `title` additionally accepts `--model` to override the model used to resolve the config before generating the title. ### `docker-agent completion` From 11f568a29cfd065eb7f885d7f893c8d538823cae Mon Sep 17 00:00:00 2001 From: David Gageot Date: Thu, 1 Oct 2026 10:24:21 +0200 Subject: [PATCH 2/2] fix: debug tool stops acting on canceled context, blocks background jobs callDebugTool now checks ctx.Err() before and after invoking a handler so a canceled/expired debug call can't still fire a tool or report success. Background jobs toolset also rejects launches when the host marks the context, since debug tool exits before any backgrounded process could ever be reaped. Assisted-By: docker-agent --- cmd/root/debug.go | 3 +- cmd/root/debug_tool.go | 9 +- cmd/root/debug_tool_test.go | 104 ++++++++++++++++++ docs/features/cli/index.md | 2 +- .../builtin/backgroundjobs/backgroundjobs.go | 3 + .../backgroundjobs/backgroundjobs_test.go | 26 +++++ pkg/tools/builtin/backgroundjobs/context.go | 10 ++ 7 files changed, 154 insertions(+), 3 deletions(-) create mode 100644 pkg/tools/builtin/backgroundjobs/context.go diff --git a/cmd/root/debug.go b/cmd/root/debug.go index 9d35c59109..e133f9dff3 100644 --- a/cmd/root/debug.go +++ b/cmd/root/debug.go @@ -91,7 +91,8 @@ func newDebugCmd() *cobra.Command { "Parameters must be a JSON object (defaults to {}). Use --agent to select an agent.\n" + "Use 'debug toolsets --json' to inspect tool names and parameter schemas.\n\n" + "Calls have real side effects and bypass session hooks and approval checks.\n" + - "Tools that require an agent runtime are not supported.", + "Tools that require an agent runtime are not supported. Built-in background jobs\n" + + "cannot be launched because toolsets are stopped when the command exits.", Example: ` docker agent debug tool agent.yaml read_file '{"path":"README.md"}' docker agent debug tool agent.yaml shell '{"cmd":"pwd"}' --agent root --json`, Args: cobra.RangeArgs(2, 3), diff --git a/cmd/root/debug_tool.go b/cmd/root/debug_tool.go index d5fa973317..1731c0185e 100644 --- a/cmd/root/debug_tool.go +++ b/cmd/root/debug_tool.go @@ -11,10 +11,11 @@ import ( "github.com/docker/docker-agent/pkg/telemetry" "github.com/docker/docker-agent/pkg/tools" + "github.com/docker/docker-agent/pkg/tools/builtin/backgroundjobs" ) func (f *debugFlags) runDebugToolCommand(cmd *cobra.Command, args []string) (commandErr error) { - ctx := cmd.Context() + ctx := backgroundjobs.WithoutBackgroundJobs(cmd.Context()) // Tool parameters may contain secrets; keep them out of command telemetry. telemetry.TrackCommand(ctx, "debug", []string{"tool"}) defer func() { @@ -91,10 +92,16 @@ func callDebugTool(ctx context.Context, tool tools.Tool, arguments string) (*too Arguments: arguments, }, } + if err := ctx.Err(); err != nil { + return nil, err + } result, err := tool.Handler(ctx, toolCall, tools.NopRuntime{}) if err != nil { return nil, fmt.Errorf("calling tool %q: %w", tool.Name, err) } + if err := ctx.Err(); err != nil { + return nil, err + } if result == nil { return nil, fmt.Errorf("tool %q returned no result", tool.Name) } diff --git a/cmd/root/debug_tool_test.go b/cmd/root/debug_tool_test.go index 0a30337a4e..cd7372912e 100644 --- a/cmd/root/debug_tool_test.go +++ b/cmd/root/debug_tool_test.go @@ -10,7 +10,9 @@ import ( "net/http/httptest" "os" "path/filepath" + "slices" "testing" + "time" "github.com/modelcontextprotocol/go-sdk/mcp" "github.com/spf13/cobra" @@ -19,6 +21,7 @@ import ( "github.com/docker/docker-agent/pkg/environment" "github.com/docker/docker-agent/pkg/tools" + "github.com/docker/docker-agent/pkg/tools/builtin/filesystem" ) const debugToolConfig = ` @@ -228,6 +231,48 @@ func TestCallDebugTool(t *testing.T) { assert.Same(t, expected, result) }) + t.Run("canceled before execution", func(t *testing.T) { + t.Parallel() + + ctx, cancel := context.WithCancel(t.Context()) + cancel() + tool := tools.Tool{Name: "test", Handler: func(context.Context, tools.ToolCall, tools.Runtime) (*tools.ToolCallResult, error) { + t.Fatal("canceled handler must not run") + return nil, nil + }} + result, err := callDebugTool(ctx, tool, "{}") + require.ErrorIs(t, err, context.Canceled) + assert.Nil(t, result) + }) + + t.Run("canceled during execution", func(t *testing.T) { + t.Parallel() + + ctx, cancel := context.WithCancel(t.Context()) + defer cancel() + tool := tools.Tool{Name: "test", Handler: func(context.Context, tools.ToolCall, tools.Runtime) (*tools.ToolCallResult, error) { + cancel() + return tools.ResultSuccess("ignored cancellation"), nil + }} + result, err := callDebugTool(ctx, tool, "{}") + require.ErrorIs(t, err, context.Canceled) + assert.Nil(t, result) + }) + + t.Run("expired deadline", func(t *testing.T) { + t.Parallel() + + ctx, cancel := context.WithDeadline(t.Context(), time.Now().Add(-time.Second)) + defer cancel() + tool := tools.Tool{Name: "test", Handler: func(context.Context, tools.ToolCall, tools.Runtime) (*tools.ToolCallResult, error) { + t.Fatal("expired handler must not run") + return nil, nil + }} + result, err := callDebugTool(ctx, tool, "{}") + require.ErrorIs(t, err, context.DeadlineExceeded) + assert.Nil(t, result) + }) + t.Run("handler error", func(t *testing.T) { t.Parallel() @@ -342,3 +387,62 @@ func TestDebugToolsetsCommand_IncludesDeferredTools(t *testing.T) { } t.Fatal("deferred shell tool must be discoverable") } + +func TestCallDebugTool_CanceledWriteFile(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + available, err := filesystem.New(dir).Tools(t.Context()) + require.NoError(t, err) + index := slices.IndexFunc(available, func(tool tools.Tool) bool { return tool.Name == filesystem.ToolNameWriteFile }) + require.NotEqual(t, -1, index) + ctx, cancel := context.WithCancel(t.Context()) + cancel() + _, err = callDebugTool(ctx, available[index], `{"path":"marker","content":"must not be written"}`) + require.ErrorIs(t, err, context.Canceled) + assert.NoFileExists(t, filepath.Join(dir, "marker")) +} + +func TestDebugToolCommand_BackgroundJobs(t *testing.T) { + t.Parallel() + + for _, mode := range []string{"direct", "deferred", "code mode"} { + t.Run(mode, func(t *testing.T) { + t.Parallel() + + for _, recall := range []bool{false, true} { + t.Run(fmt.Sprintf("recall=%t", recall), func(t *testing.T) { + t.Parallel() + + flags := &debugFlags{} + flags.runConfig.GlobalCodeMode = mode == "code mode" + cfg := fmt.Sprintf(`agents: + root: + model: test + toolsets: + - type: background_jobs + recall: %t + defer: %t +models: + test: + provider: openai + model: gpt-4o + max_tokens: 100 +`, recall, mode == "deferred") + args := []string{"run_background_job", `{"cmd":"echo must-not-run"}`} + if mode == "code mode" { + args = []string{"run_tools_with_javascript", `{"script":"try { await run_background_job({cmd: 'echo must-not-run'}); return 'unexpected success'; } catch (error) { return error.message; }"}`} + } + out, err := runDebugToolConfig(t, flags, cfg, args...) + if mode == "code mode" { + require.NoError(t, err) + assert.Contains(t, out, "background jobs are not supported by this host") + } else { + require.ErrorContains(t, err, "background jobs are not supported by this host") + assert.Empty(t, out) + } + }) + } + }) + } +} diff --git a/docs/features/cli/index.md b/docs/features/cli/index.md index 57626f167c..4e63e67a66 100644 --- a/docs/features/cli/index.md +++ b/docs/features/cli/index.md @@ -777,7 +777,7 @@ $ docker agent debug oauth login agent.yaml github > [!WARNING] > **`debug tool` executes real tool calls** > -> Calls can modify files, run commands, or contact external services. They bypass session hooks and approval checks. Tools that require an agent runtime (such as delegation and handoff) are not supported. Tool errors print their result and exit with a non-zero status. Use `debug toolsets --json` to inspect parameter schemas before calling a tool. +> Calls can modify files, run commands, or contact external services. They bypass session hooks and approval checks. Tools that require an agent runtime (such as delegation and handoff) are not supported. Built-in background jobs cannot be launched because toolsets are stopped when the command exits; use `shell` for synchronous commands instead. Tool errors print their result and exit with a non-zero status. Use `debug toolsets --json` to inspect parameter schemas before calling a tool. > [!WARNING] > **`debug auth --json` prints the full bearer token** diff --git a/pkg/tools/builtin/backgroundjobs/backgroundjobs.go b/pkg/tools/builtin/backgroundjobs/backgroundjobs.go index 4e6c85db26..0d571d3748 100644 --- a/pkg/tools/builtin/backgroundjobs/backgroundjobs.go +++ b/pkg/tools/builtin/backgroundjobs/backgroundjobs.go @@ -228,6 +228,9 @@ func (h *backgroundJobsHandler) RunBackgroundJobWithRecall(ctx context.Context, } func (h *backgroundJobsHandler) runBackgroundJob(ctx context.Context, rt tools.Runtime, params runBackgroundJobParams) (*tools.ToolCallResult, error) { + if blocked, _ := ctx.Value(noBackgroundJobsKey{}).(bool); blocked { + return nil, errors.New("background jobs are not supported by this host; use the shell tool for synchronous commands") + } if strings.TrimSpace(params.Cmd) == "" { return tools.ResultError(`Error: missing or empty "cmd" parameter. Pass the shell command as {"cmd": "..."}.`), nil } diff --git a/pkg/tools/builtin/backgroundjobs/backgroundjobs_test.go b/pkg/tools/builtin/backgroundjobs/backgroundjobs_test.go index 02dd1ef4f2..87b865cbbf 100644 --- a/pkg/tools/builtin/backgroundjobs/backgroundjobs_test.go +++ b/pkg/tools/builtin/backgroundjobs/backgroundjobs_test.go @@ -573,3 +573,29 @@ func TestBackgroundJobsTool_BackgroundedChildDoesNotBlockReturn(t *testing.T) { assert.Contains(t, listResult.Output, "Status: completed") assert.Contains(t, listResult.Output, "Exit Code: 0") } + +func TestBackgroundJobsTool_WithoutBackgroundJobs(t *testing.T) { + t.Parallel() + + for _, recall := range []bool{false, true} { + t.Run(map[bool]string{false: "no recall", true: "recall"}[recall], func(t *testing.T) { + t.Parallel() + + toolset := newTestTool(t) + toolset.handler.recall = recall + available, err := toolset.Tools(t.Context()) + require.NoError(t, err) + ctx, cancel := context.WithCancel(WithoutBackgroundJobs(t.Context())) + defer cancel() + result, err := available[0].Handler(ctx, tools.ToolCall{ + Function: tools.FunctionCall{ + Name: ToolNameRunBackgroundJob, + Arguments: `{"cmd":"echo must-not-run","recall":true}`, + }, + }, tools.NopRuntime{}) + require.ErrorContains(t, err, "background jobs are not supported by this host") + assert.Nil(t, result) + assert.Zero(t, toolset.handler.jobCounter.Load(), "rejection must precede spawning a process") + }) + } +} diff --git a/pkg/tools/builtin/backgroundjobs/context.go b/pkg/tools/builtin/backgroundjobs/context.go new file mode 100644 index 0000000000..85f3c3804d --- /dev/null +++ b/pkg/tools/builtin/backgroundjobs/context.go @@ -0,0 +1,10 @@ +package backgroundjobs + +import "context" + +type noBackgroundJobsKey struct{} + +// WithoutBackgroundJobs prevents job launches in hosts that stop their toolsets after each call. +func WithoutBackgroundJobs(ctx context.Context) context.Context { + return context.WithValue(ctx, noBackgroundJobsKey{}, true) +}